<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e94904</article-id><article-id pub-id-type="doi">10.2196/94904</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Incremental Diagnostic Value of Clinical Information for Large Language Models Across Multiple Organs: Retrospective Study</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Zhang</surname><given-names>Jinqi</given-names></name><degrees>BSc</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Wang</surname><given-names>Xiaoyi</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhao</surname><given-names>Yanfeng</given-names></name><degrees>MM</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Cui</surname><given-names>Xiaolin</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhao</surname><given-names>Liyun</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhao</surname><given-names>Xinming</given-names></name><degrees>MPA</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Hongmei</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Zhu</surname><given-names>Zheng</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Diagnostic Radiology, National Cancer Center/National Clinical Research Center for Cancer/Cancer Hospital, Chinese Academy of Medical Sciences &#x0026; Peking Union Medical College</institution><addr-line>No. 17, Panjiayuan Nanli, Chaoyang District</addr-line><addr-line>Beijing</addr-line><country>China</country></aff><aff id="aff2"><institution>Department of Radiology, Beijing Friendship Hospital, Capital Medical University</institution><addr-line>Beijing</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Yoshikawa</surname><given-names>Takeharu</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Kikuchi</surname><given-names>Tomohiro</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Zheng Zhu, MD, Department of Diagnostic Radiology, National Cancer Center/National Clinical Research Center for Cancer/Cancer Hospital, Chinese Academy of Medical Sciences &#x0026; Peking Union Medical College, No. 17, Panjiayuan Nanli, Chaoyang District, Beijing, 100021, China, 86 13520102729; <email>dr_zhuzheng@sina.com</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>26</day><month>8</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e94904</elocation-id><history><date date-type="received"><day>12</day><month>03</month><year>2026</year></date><date date-type="rev-recd"><day>21</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>23</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Jinqi Zhang, Xiaoyi Wang, Yanfeng Zhao, Xiaolin Cui, Liyun Zhao, Xinming Zhao, Hongmei Zhang, Zheng Zhu. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 26.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e94904"/><abstract><sec><title>Background</title><p>Although large language models (LLMs) have demonstrated the ability to generate the impression section from radiology findings automatically, the incremental diagnostic value of clinical information for these models remains unclear.</p></sec><sec><title>Objective</title><p>This study aimed to evaluate the incremental diagnostic value of clinical information for LLMs and compare their performance with that of radiologists.</p></sec><sec sec-type="methods"><title>Methods</title><p>This retrospective study included radiology reports from patients with histopathologically confirmed liver, lung, and breast diseases from 2 institutions between October 2021 and February 2025. We defined three progressive information input scenarios: (1) basic patient information and imaging findings, (2) scenario A plus chief complaint or clinical history, and (3) scenario B plus key laboratory results. Scenario-based data were input into 3 general-purpose LLMs (DeepSeek-R1, Gemini 2.5 Pro, and GPT-4o), generating 2709 entries. Diagnostic accuracy was assessed for both benign-malignant differentiation and disease diagnosis, with histopathology serving as the reference standard. Accuracy was compared among scenarios and against radiologist performance using the McNemar test, and <italic>P</italic> values were adjusted using the Holm-Bonferroni correction for multiple comparisons.</p></sec><sec sec-type="results"><title>Results</title><p>A total of 301 patients with pathologically confirmed diseases were included (mean age 53.5, SD 12.0 years; women: n=208, 69.1%). In the liver cohort, a numerical trend toward higher accuracy was observed in scenario C compared with scenario A across all 3 models (scenario C range: 72.3%&#x2010;76.2% vs scenario A range: 64.4%&#x2010;68.3%); these differences did not reach statistical significance after Holm-Bonferroni correction (all adjusted <italic>P</italic>&#x003E;.99). Notably, the DeepSeek-R1 model in scenario C achieved the highest diagnostic accuracy (77/101, 76.2%), with no evidence of a difference compared with radiologists (82/101, 81.2%; adjusted <italic>P</italic>&#x003E;.99). In contrast, results in the lung and breast cohorts were more heterogeneous. In the lung cohort, GPT-4o achieved its highest accuracy in scenario A for disease diagnosis (68/92, 73.9%), which exceeded its performance in scenario B (64/92, 69.6%) and scenario C (66/92, 71.7%), suggesting that additional clinical information did not confer a consistent benefit. Gemini 2.5 Pro in scenario B achieved the highest accuracy in this cohort (72/92, 78.3%); however, no statistically significant difference was found compared with radiologists (80/92, 87.0%; adjusted <italic>P</italic>=.25). In the breast cohort, DeepSeek-R1 achieved the numerically highest diagnostic accuracy in scenario A, and it decreased numerically with the addition of laboratory tests, although no significant difference was found between scenarios A and C (73/108, 67.6% vs 71/108, 65.7%; adjusted <italic>P</italic>&#x003E;.99).</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>While the addition of clinical information was associated with a numeric trend toward higher diagnostic accuracy overall, this trend was heterogeneous across models and disease types, and no statistically significant improvement was demonstrated after adjustment for multiple comparisons.</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>diagnostic accuracy</kwd><kwd>clinical history</kwd><kwd>radiology report</kwd><kwd>radiologist</kwd><kwd>ChatGPT</kwd><kwd>DeepSeek</kwd><kwd>Gemini</kwd><kwd>laboratory investigation</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>With the increasing prevalence of large language models (LLMs), their application in radiology has expanded, offering potential benefits to both radiologists and patients. Current applications include evaluating medical knowledge on board examinations [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>], simplifying radiology reports to improve patient readability [<xref ref-type="bibr" rid="ref4">4</xref>-<xref ref-type="bibr" rid="ref6">6</xref>], addressing patient inquiries [<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref10">10</xref>], and extracting data from radiology reports [<xref ref-type="bibr" rid="ref11">11</xref>-<xref ref-type="bibr" rid="ref15">15</xref>]. The process of generating a diagnostic report typically comprises two steps: (1) identifying radiologic findings and (2) synthesizing these findings with clinical context to formulate the final impression. Zhou et al [<xref ref-type="bibr" rid="ref16">16</xref>] evaluated the feasibility of using GPT-4V (OpenAI) to detect radiologic findings on chest radiographs and reported limited diagnostic accuracy for this initial step. Conversely, GPT-4 (OpenAI) has demonstrated the potential to detect errors in radiology reports [<xref ref-type="bibr" rid="ref17">17</xref>]. Furthermore, it can use logical reasoning to resolve discrepancies [<xref ref-type="bibr" rid="ref18">18</xref>], thereby enhancing radiologists&#x2019; efficiency. Regarding the second step, LLMs have demonstrated the ability to generate the impression section from radiology findings automatically [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref21">21</xref>] and to propose differential diagnoses based on imaging findings [<xref ref-type="bibr" rid="ref22">22</xref>]. Sun et al [<xref ref-type="bibr" rid="ref19">19</xref>] noted that GPT-4&#x2019;s capabilities in zero-shot impression generation from radiology report findings remained largely unexplored and conducted preliminary research. They compared impressions generated by radiologists and GPT-4 across multiple dimensions, demonstrating GPT-4&#x2019;s proficiency in linguistic coherence. Building on this work, Zhang et al [<xref ref-type="bibr" rid="ref20">20</xref>] conducted further research by constructing a specialized LLM for radiology through pre-training and fine-tuning, extending report impression generation to comprehensive imaging modalities and anatomical regions. They found that while medical-specific models, such as WiNGPT, performed well across most dimensions, they significantly underperformed in providing specific diagnoses. Mukherjee et al [<xref ref-type="bibr" rid="ref23">23</xref>] found that LLMs exhibit a stronger dependence on text than on images. Recently, several studies have begun to explore the incremental diagnostic value of clinical information. For instance, evaluations using the &#x201C;Diagnosis Please&#x201D; quizzes from radiology and pediatric case descriptions have demonstrated that integrating imaging findings with clinical history significantly improves the diagnostic accuracy of models such as GPT-4 and Claude 3.5 Sonnet [<xref ref-type="bibr" rid="ref24">24</xref>-<xref ref-type="bibr" rid="ref26">26</xref>]. Similarly, incorporating clinical information into the process of generating impressions from brain magnetic resonance imaging (MRI) findings has markedly enhanced model performance [<xref ref-type="bibr" rid="ref27">27</xref>].</p><p>Meanwhile, LLM technology is rapidly advancing from closed-source families such as GPT to high-performance open-source models and specialized architectures that support local deployment [<xref ref-type="bibr" rid="ref28">28</xref>-<xref ref-type="bibr" rid="ref30">30</xref>]. However, these previous studies have primarily relied on rigorously curated textbook quizzes, focused on specific subspecialties, or evaluated only a limited number of models.</p><p>Therefore, the purpose of this study was to systematically evaluate the diagnostic accuracy of 3 general-purpose LLMs (DeepSeek-R1 [DeepSeek], Gemini 2.5 Pro [Google], and GPT-4o) in assessing liver, lung, and breast diseases across 3 progressive clinical information input scenarios.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Ethical Considerations</title><p>The study protocol was reviewed and approved by the Ethics Committee of the Cancer Hospital of the Chinese Academy of Medical Sciences (approval 25/331&#x2010;5277). The Ethics Committee waived the requirement for informed consent, determining that the retrospective use of deidentified data posed no risk to patients. To ensure confidentiality, all patient identifiers were removed prior to analysis. No images of individual participants are included in this manuscript or its supplementary materials.</p></sec><sec id="s2-2"><title>Research Data</title><p>Data for this study were obtained from 2 hospitals. Radiology reports were retrieved from the hospitals&#x2019; radiological information systems from October 2021 to February 2025, and consecutive cases were collected and comprised 101 liver, 92 lung, and 108 breast cases. Inclusion criteria were as follows: patients (1) aged &#x2265;18 years; (2) who underwent contrast-enhanced liver MRI, chest computed tomography (CT), or breast MRI for disease screening or diagnostic purposes; and (3) who underwent surgical resection with histopathologic confirmation. Exclusion criteria were as follows: (1) lack of relevant preoperative laboratory results and (2) incomplete radiologic findings or impression text. For patients with multiple preoperative examinations, only the most recent scan before surgical resection was selected to ensure the closest temporal proximity to the pathological diagnosis and to maintain the independence of the observations.</p></sec><sec id="s2-3"><title>Data Collection and Preparation</title><p>We collected patient radiology reports containing both findings and impression sections, along with clinical information such as demographics, chief complaints, and key laboratory test results obtained prior to histopathologic examination. Then, we defined three progressive information input scenarios: (1) basic patient information and imaging findings, (2) scenario A plus the chief complaint or clinical history (only the chief complaint was available for liver and lung cases), and (3) scenario B plus key laboratory test results relevant to each disease. All patient-identifying information was anonymized. We then populated the collected data according to these 3 scenarios and generated separate templates for each disease. To mimic the real-world clinical context, the original text (in Chinese) was kept unchanged during input. Examples of templates for different diseases are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. All model queries were conducted via the public consumer web interfaces of the evaluated models. These interfaces do not support separate system-level instructions; all role definitions, task descriptions, and formatting constraints were therefore incorporated directly into the user input templates provided in the supplementary materials. All clinical text was manually deidentified prior to input, with patient names, medical record numbers, and institutional identifiers removed. The option to decline data use for model training was selected in the data management settings. No content beyond the deidentified text shown in the appendix templates was entered.</p><p>It is important to note that while the LLMs were tested across 3 progressive experimental scenarios to evaluate the incremental value of additional data, the radiologist&#x2019;s performance remained a fixed historical benchmark and was not retested under the same constrained scenarios.</p><p>All radiology reports used as the historical baseline were authored by experienced attending radiologists specializing in abdominal, thoracic, or breast imaging, and detailed demographic and experience characteristics are provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p><p>All radiology reports and clinical data used in this study were sourced exclusively from the institutional hospital information system with strictly controlled access. None of these records were publicly released or indexed at any time, and therefore, they could not have been included in the pre-training data of any evaluated model.</p><p>To minimize bias arising from the models&#x2019; context sensitivity and to avoid memory interference, chat sessions were restarted for each model, and responses were collected after each report was typed, yielding a total of 2709 entries.</p></sec><sec id="s2-4"><title>Report Processing and Evaluation</title><p>The LLM-generated reports were exported to Microsoft Excel. A binary semantic matching rubric was used to compare model outputs against the histopathologic reference standard. A case was deemed consistent (score 1) if the core diagnosis matched the reference, regardless of minor stylistic variation; it was deemed inconsistent (score 0) if the diagnosis contradicted the reference or no definitive diagnosis was given. For all models, only the final diagnostic output was evaluated; the internal reasoning traces generated by DeepSeek-R1 were not considered in the scoring process. All cases were independently evaluated by 2 raters: a radiologist (XW, with 23 y of experience) and a medical student (JZ, with 2 y of experience). Disagreements were resolved through consensus with a third evaluator, a radiologist (ZZ, with 21 y of experience). To assess the reliability of the free-text evaluation, interrater agreement between the 2 primary evaluators was measured using Cohen &#x03BA; coefficient. The &#x03BA; value and its 95% CI are reported in Table S1 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p><p>For evaluation of the breast cohort, reports that included a specific diagnostic name were classified according to the established biological nature of that entity, irrespective of the Breast Imaging Reporting and Data System (BI-RADS) score. For reports without a specific diagnosis, a strict BI-RADS&#x2013;based threshold was applied: categories 1 to 3 were defined as benign, and categories 4a to 6 were defined as malignant. For the purpose of determining diagnostic consistency, cases were classified as inconsistent if the original radiology report provided only a risk classification without explicitly stating a specific histologic diagnosis in the impression section. Per the classification rubric described earlier, any case deemed inconsistent was counted as an incorrect diagnosis and was not excluded from the denominator.</p></sec><sec id="s2-5"><title>LLMs</title><p>We used the 3 generic LLM versions GPT-4o (accessed July 11 to August 9, 2025), Gemini 2.5 Pro (accessed July 17 to August 9, 2025), and DeepSeek-R1 (accessed July 13-31, 2025). All model queries were performed via the public consumer web interfaces of the evaluated models. No specific hyperparameter values were manually set or configured during this process. For reference, the default hyperparameter values reported in the official documentation for each model are listed in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. The LLMs were tasked with generating radiologic impressions from the input data defined in the various scenarios.</p></sec><sec id="s2-6"><title>Study Design</title><p>We compared the diagnostic accuracy of LLMs with that of radiologists for &#x201C;benign-malignant differentiation&#x201D; and &#x201C;disease diagnosis.&#x201D; This evaluation was conducted using two reference standards: (1) for the entire study cohort, histopathologic diagnosis served as the reference standard; and (2) for the breast cohort specifically, the final impressions served as the reference standard to evaluate model performance in BI-RADS classification. We separately recorded the diagnostic accuracy of each model for the benign and malignant subgroups. Furthermore, we adhered to the principle of single-variable analysis to independently assess the effects of model architecture, input scenario, and disease type on diagnostic performance.</p></sec><sec id="s2-7"><title>Statistical Analyses</title><p>Data were analyzed using R software (version 4.4.2; R Foundation for Statistical Computing). To address the imbalanced nature of the cohorts, diagnostic performance for benign-malignant differentiation was evaluated comprehensively using overall accuracy, sensitivity, specificity, positive predictive value, and negative predictive value. Pairwise comparisons of diagnostic accuracy were performed using the McNemar test. To control the cumulative risk of type I errors (false positives) arising from multiple pairwise comparisons, the Holm-Bonferroni correction was applied to all <italic>P</italic> values derived from the McNemar tests. The correction was applied separately within each diagnostic task for each organ cohort, so that the family-wise error rate was controlled independently for each analysis unit. Statistical significance was defined as an adjusted <italic>P</italic>&#x003C;.05.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Study Sample</title><p>A total of 301 patients with pathologically confirmed diseases were included in the study (liver: n=101, 33.5%; lung: n=92, 30.6%; and breast: n=108, 35.9%). The flowchart of patient inclusion and exclusion and the diagram of the study design are shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>. The mean patient age was 55.1 (SD 12.2) years for the liver cohort, 54.5 (SD 12.3) years for the lung cohort, and 51.1 (SD 11.4) years for the breast cohort. There were 38 (37.6%), 62 (67.4%), and 108 (100.0%) female patients in the liver, lung, and breast cohorts, respectively. Detailed patient demographics and pathologic characteristics are summarized in <xref ref-type="table" rid="table1">Table 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Study flowcharts. (A) Flowchart of patient inclusion and exclusion. (B, C) Study design workflow. We defined 3 progressive information input scenarios: A, basic patient information and imaging findings; B, scenario A plus chief complaint or clinical history; and C, scenario B plus key laboratory results. BI-RADS: Breast Imaging Reporting and Data System; ChatGPT: GPT-4o; CT: computed tomography; DeepSeek: DeepSeek-R1; Gemini: Gemini 2.5 Pro; LLM: large language model; MRI: magnetic resonance imaging.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e94904_fig01.png"/></fig><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Patient demographics and histopathologic reference standards.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Characteristic</td><td align="left" valign="top">Liver (n=101)</td><td align="left" valign="top">Lung (n=92)</td><td align="left" valign="top">Breast (n=108)</td></tr></thead><tbody><tr><td align="left" valign="top">Age (y), mean (SD)</td><td align="char" char="." valign="top">55.1 (12.2)</td><td align="char" char="." valign="top">54.5 (12.3)</td><td align="char" char="." valign="top">51.1 (11.4)</td></tr><tr><td align="left" valign="top">Sex, n (%)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Female</td><td align="char" char="." valign="top">38 (37.6)</td><td align="char" char="." valign="top">62 (67.4)</td><td align="char" char="." valign="top">108 (100.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Male</td><td align="char" char="." valign="top">63 (62.4)</td><td align="char" char="." valign="top">30 (32.6)</td><td align="char" char="." valign="top">0 (0.0)</td></tr><tr><td align="left" valign="top">Lesion characterization, n (%)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Benign</td><td align="char" char="." valign="top">32 (31.7)</td><td align="char" char="." valign="top">24 (26.1)</td><td align="char" char="." valign="top">32 (29.6)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Malignant</td><td align="char" char="." valign="top">69 (68.3)</td><td align="char" char="." valign="top">68 (73.9)</td><td align="char" char="." valign="top">76 (70.4)</td></tr><tr><td align="left" valign="top">Histopathologic diagnosis, n (%)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>HCC<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="char" char="." valign="top">48 (47.5)</td><td align="char" char="." valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LUAD<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">&#x2014;</td><td align="char" char="." valign="top">66 (71.7)</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>IBC<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="char" char="." valign="top">68 (63.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hemangioma</td><td align="char" char="." valign="top">14 (13.9)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>PSP<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup></td><td align="left" valign="top">&#x2014;</td><td align="char" char="." valign="top">14 (15.2)</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FA<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup></td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="char" char="." valign="top">21 (19.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Other<sup><xref ref-type="table-fn" rid="table1fn7">g</xref></sup></td><td align="char" char="." valign="top">39 (38.6)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Other<sup><xref ref-type="table-fn" rid="table1fn8">h</xref></sup></td><td align="left" valign="top">&#x2014;</td><td align="char" char="." valign="top">12 (13.0)</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Other<sup><xref ref-type="table-fn" rid="table1fn9">i</xref></sup></td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="char" char="." valign="top">19 (17.6)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>HCC: hepatocellular carcinoma.</p></fn><fn id="table1fn2"><p><sup>b</sup>Not applicable.</p></fn><fn id="table1fn3"><p><sup>c</sup>LUAD: lung adenocarcinoma.</p></fn><fn id="table1fn4"><p><sup>d</sup>IBC: invasive breast carcinoma.</p></fn><fn id="table1fn5"><p><sup>e</sup>PSP: pulmonary sclerosing pneumocytoma.</p></fn><fn id="table1fn6"><p><sup>f</sup>FA: fibroadenoma.</p></fn><fn id="table1fn7"><p><sup>g</sup>Other liver pathologies include cholangiocarcinoma, liver metastasis, hepatic adenoma, focal nodular hyperplasia, angiomyolipoma, low-grade intraepithelial neoplasia, pleomorphic undifferentiated sarcoma, hepatic perivascular epithelioid cell tumor, combined hepatocellular cholangiocarcinoma, and epithelioid hemangioendothelioma.</p></fn><fn id="table1fn8"><p><sup>h</sup>Other lung pathologies include squamous cell carcinoma, atypical adenomatous hyperplasia, pulmonary lymphangioma, and bronchiolar adenoma.</p></fn><fn id="table1fn9"><p><sup>i</sup>Other breast pathologies include ductal carcinoma in situ, intraductal papilloma, lipoma, adenomyoma, and malignant phyllodes tumor.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Comparison of Diagnostic Accuracy in Different Scenarios</title><sec id="s3-2-1"><title>Overview</title><p>The accuracy of the models for benign-malignant differentiation and disease diagnosis across the 3 scenarios is presented in <xref ref-type="table" rid="table2">Tables 2</xref> and <xref ref-type="table" rid="table3">3</xref> and <xref ref-type="fig" rid="figure2">Figure 2</xref>. Overall, the 3 LLMs demonstrated consistent performance trends across liver, lung, and breast diseases. Diagnostic accuracy was higher for benign-malignant differentiation (range 71.7%&#x2010;88.1%) than for disease diagnosis (range 50%&#x2010;78.7%). Given the consistency of these trends, the subsequent results focus on disease diagnosis. All diagnostic performance metric values are presented in Table S2 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>. Additionally, we performed McNemar tests for all combinations of models and scenarios; the <italic>P</italic> value matrices for all pairwise comparisons are provided in Table S3 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Diagnostic accuracy of large language models (LLMs) versus radiologists for benign-malignant differentiation<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Organ and scenario</td><td align="left" valign="top" colspan="2">Radiologist</td><td align="left" valign="top" colspan="2">DeepSeek<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td><td align="left" valign="top" colspan="2">Gemini<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td><td align="left" valign="top" colspan="2">GPT<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="top" colspan="3"><italic>P</italic> value<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup></td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">n (%)</td><td align="left" valign="top">95% CI</td><td align="left" valign="top">n (%)</td><td align="left" valign="top">95% CI</td><td align="left" valign="top">n (%)</td><td align="left" valign="top">95% CI</td><td align="left" valign="top">n (%)</td><td align="left" valign="top">95% CI</td><td align="left" valign="top">DeepSeek versus radiologist</td><td align="left" valign="top">Gemini versus radiologist</td><td align="left" valign="top">GPT versus radiologist</td></tr></thead><tbody><tr><td align="left" valign="top">Liver (n=101)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>a<sup><xref ref-type="table-fn" rid="table2fn6">f</xref></sup></td><td align="char" char="." valign="top">91 (90.1)</td><td align="char" char="." valign="top">82.1&#x2010;94.9</td><td align="char" char="." valign="top">85 (84.2)</td><td align="char" char="." valign="top">75.2&#x2010;90.4</td><td align="char" char="." valign="top">84 (83.2)</td><td align="char" char="." valign="top">74.1&#x2010;89.6</td><td align="char" char="." valign="top">78 (77.2)</td><td align="char" char="." valign="top">67.6&#x2010;84.7</td><td align="char" char="." valign="top">.15</td><td align="char" char="." valign="top">.10</td><td align="char" char="." valign="top">.002</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>b<sup><xref ref-type="table-fn" rid="table2fn7">g</xref></sup></td><td align="char" char="." valign="top">91 (90.1)</td><td align="char" char="." valign="top">82.1&#x2010;94.9</td><td align="char" char="." valign="top">89 (88.1)</td><td align="char" char="." valign="top">79.8&#x2010;93.4</td><td align="char" char="." valign="top">82 (81.2)</td><td align="char" char="." valign="top">71.9&#x2010;88.0</td><td align="char" char="." valign="top">82 (81.2)</td><td align="char" char="." valign="top">71.9&#x2010;88.0</td><td align="char" char="." valign="top">.75</td><td align="char" char="." valign="top">.02</td><td align="char" char="." valign="top">.03</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>c<sup><xref ref-type="table-fn" rid="table2fn8">h</xref></sup></td><td align="char" char="." valign="top">91 (90.1)</td><td align="char" char="." valign="top">82.1&#x2010;94.9</td><td align="char" char="." valign="top">89 (88.1)</td><td align="char" char="." valign="top">79.8&#x2010;93.4</td><td align="char" char="." valign="top">85 (84.2)</td><td align="char" char="." valign="top">75.2&#x2010;90.4</td><td align="char" char="." valign="top">86 (85.1)</td><td align="char" char="." valign="top">76.4&#x2010;91.2</td><td align="char" char="." valign="top">.75</td><td align="char" char="." valign="top">.11</td><td align="char" char="." valign="top">.27</td></tr><tr><td align="left" valign="top">Lung (n=92)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>a</td><td align="char" char="." valign="top">80 (87.0)</td><td align="char" char="." valign="top">77.9&#x2010;92.8</td><td align="char" char="." valign="top">67 (72.8)</td><td align="char" char="." valign="top">62.4&#x2010;81.3</td><td align="char" char="." valign="top">70 (76.1)</td><td align="char" char="." valign="top">65.9&#x2010;84.1</td><td align="char" char="." valign="top">69 (75.0)</td><td align="char" char="." valign="top">64.7&#x2010;83.2</td><td align="char" char="." valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table2fn9">i</xref></sup></td><td align="char" char="." valign="top">.004</td><td align="char" char="." valign="top">.003</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>b</td><td align="char" char="." valign="top">80 (87.0)</td><td align="char" char="." valign="top">77.9&#x2010;92.8</td><td align="char" char="." valign="top">71 (77.2)</td><td align="char" char="." valign="top">67.0&#x2010;85.0</td><td align="char" char="." valign="top">73 (79.3)</td><td align="char" char="." valign="top">69.4&#x2010;86.8</td><td align="char" char="." valign="top">67 (72.8)</td><td align="char" char="." valign="top">62.4&#x2010;81.3</td><td align="char" char="." valign="top">.008</td><td align="char" char="." valign="top">.02</td><td align="char" char="." valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table2fn9">i</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>c</td><td align="char" char="." valign="top">80 (87.0)</td><td align="char" char="." valign="top">77.9&#x2010;92.8</td><td align="char" char="." valign="top">69 (75.0)</td><td align="char" char="." valign="top">64.7&#x2010;83.2</td><td align="char" char="." valign="top">66 (71.7)</td><td align="char" char="." valign="top">61.2&#x2010;80.4</td><td align="char" char="." valign="top">67 (72.8)</td><td align="char" char="." valign="top">62.4&#x2010;81.3</td><td align="char" char="." valign="top">.01</td><td align="char" char="." valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table2fn9">i</xref></sup></td><td align="char" char="." valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table2fn9">i</xref></sup></td></tr><tr><td align="left" valign="top">Breast (n=108)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>a</td><td align="char" char="." valign="top">99 (91.7)</td><td align="char" char="." valign="top">84.4&#x2010;95.9</td><td align="char" char="." valign="top">85 (78.7)</td><td align="char" char="." valign="top">69.6&#x2010;85.8</td><td align="char" char="." valign="top">84 (77.8)</td><td align="char" char="." valign="top">68.6&#x2010;85.0</td><td align="char" char="." valign="top">82 (75.9)</td><td align="char" char="." valign="top">66.6&#x2010;83.4</td><td align="char" char="." valign="top">.006</td><td align="char" char="." valign="top">.004</td><td align="char" char="." valign="top">.001<sup><xref ref-type="table-fn" rid="table2fn9">i</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>b</td><td align="char" char="." valign="top">99 (91.7)</td><td align="char" char="." valign="top">84.4&#x2010;95.9</td><td align="char" char="." valign="top">87 (80.6)</td><td align="char" char="." valign="top">71.6&#x2010;87.3</td><td align="char" char="." valign="top">84 (77.8)</td><td align="char" char="." valign="top">68.6&#x2010;85.0</td><td align="char" char="." valign="top">81 (75.0)</td><td align="char" char="." valign="top">65.6&#x2010;82.6</td><td align="char" char="." valign="top">.01</td><td align="char" char="." valign="top">.004</td><td align="char" char="." valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table2fn9">i</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>c</td><td align="char" char="." valign="top">99 (91.7)</td><td align="char" char="." valign="top">84.4&#x2010;95.9</td><td align="char" char="." valign="top">85 (78.7)</td><td align="char" char="." valign="top">69.6&#x2010;85.8</td><td align="char" char="." valign="top">84 (77.8)</td><td align="char" char="." valign="top">68.6&#x2010;85.0</td><td align="char" char="." valign="top">81 (75.0)</td><td align="char" char="." valign="top">65.6&#x2010;82.6</td><td align="char" char="." valign="top">.004</td><td align="char" char="." valign="top">.004</td><td align="char" char="." valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table2fn9">i</xref></sup></td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Comparisons between LLMs and radiologists were made using the McNemar test.</p></fn><fn id="table2fn2"><p><sup>b</sup>DeepSeek: DeepSeek-R1.</p></fn><fn id="table2fn3"><p><sup>c</sup>Gemini: Gemini 2.5 Pro.</p></fn><fn id="table2fn4"><p><sup>d</sup>GPT: GPT-4o.</p></fn><fn id="table2fn5"><p><sup>e</sup><italic>P</italic> values displayed are nominal (unadjusted).</p></fn><fn id="table2fn6"><p><sup>f</sup>Basic patient information and imaging findings.</p></fn><fn id="table2fn7"><p><sup>g</sup>Scenario A plus chief complaint or clinical history.</p></fn><fn id="table2fn8"><p><sup>h</sup>Scenario B plus key laboratory results.</p></fn><fn id="table2fn9"><p><sup>i</sup>Effects that remained significant after Holm-Bonferroni correction</p></fn></table-wrap-foot></table-wrap><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Diagnostic accuracy of large language models (LLMs) versus radiologists for disease diagnosis<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Organ and scenario</td><td align="left" valign="top" colspan="2">Radiologist</td><td align="left" valign="top" colspan="2">DeepSeek<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td><td align="left" valign="top" colspan="2">Gemini<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="top" colspan="2">GPT<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top" colspan="3"><italic>P</italic> value<sup><xref ref-type="table-fn" rid="table3fn5">e</xref></sup></td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="top">n (%)</td><td align="left" valign="top">95% CI</td><td align="left" valign="top">n (%)</td><td align="left" valign="top">95% CI</td><td align="left" valign="top">n (%)</td><td align="left" valign="top">95% CI</td><td align="left" valign="top">n (%)</td><td align="left" valign="top">95% CI</td><td align="left" valign="top">DeepSeek versus radiologist</td><td align="left" valign="top">Gemini versus radiologist</td><td align="left" valign="top">GPT versus radiologist</td></tr></thead><tbody><tr><td align="left" valign="top">Liver (n=101)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>a<sup><xref ref-type="table-fn" rid="table3fn6">f</xref></sup></td><td align="char" char="." valign="top">82 (81.2)</td><td align="char" char="." valign="top">71.9&#x2010;88.0</td><td align="char" char="." valign="top">69 (68.3)</td><td align="char" char="." valign="top">58.2&#x2010;77.0</td><td align="char" char="." valign="top">69 (68.3)</td><td align="char" char="." valign="top">58.2&#x2010;77.0</td><td align="char" char="." valign="top">65 (64.4)</td><td align="char" char="." valign="top">54.1&#x2010;73.5</td><td align="char" char="." valign="top">.004</td><td align="char" char="." valign="top">.004</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table3fn7">g</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>b<sup><xref ref-type="table-fn" rid="table3fn8">h</xref></sup></td><td align="char" char="." valign="top">82 (81.2)</td><td align="char" char="." valign="top">71.9&#x2010;88.0</td><td align="char" char="." valign="top">73 (72.3)</td><td align="char" char="." valign="top">62.3&#x2010;80.5</td><td align="char" char="." valign="top">72 (71.3)</td><td align="char" char="." valign="top">61.3&#x2010;79.6</td><td align="char" char="." valign="top">70 (69.3)</td><td align="char" char="." valign="top">59.2&#x2010;77.9</td><td align="char" char="." valign="top">.02</td><td align="char" char="." valign="top">.009</td><td align="char" char="." valign="top">.003</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>c<sup><xref ref-type="table-fn" rid="table3fn9">i</xref></sup></td><td align="char" char="." valign="top">82 (81.2)</td><td align="char" char="." valign="top">71.9&#x2010;88.0</td><td align="char" char="." valign="top">77 (76.2)</td><td align="char" char="." valign="top">66.5&#x2010;83.9</td><td align="char" char="." valign="top">76 (75.2)</td><td align="char" char="." valign="top">65.5&#x2010;83.1</td><td align="char" char="." valign="top">73 (72.3)</td><td align="char" char="." valign="top">62.3&#x2010;80.5</td><td align="char" char="." valign="top">.18</td><td align="char" char="." valign="top">.08</td><td align="char" char="." valign="top">.07</td></tr><tr><td align="left" valign="top">Lung (n=92)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>a</td><td align="char" char="." valign="top">80 (87.0)</td><td align="char" char="." valign="top">77.9&#x2010;92.8</td><td align="char" char="." valign="top">63 (68.5)</td><td align="char" char="." valign="top">57.8&#x2010;77.5</td><td align="char" char="." valign="top">69 (75.0)</td><td align="char" char="." valign="top">64.7&#x2010;83.2</td><td align="char" char="." valign="top">68 (73.9)</td><td align="char" char="." valign="top">63.5&#x2010;82.3</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table3fn7">g</xref></sup></td><td align="char" char="." valign="top">.003</td><td align="left" valign="top">.002<sup><xref ref-type="table-fn" rid="table3fn7">g</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>b</td><td align="char" char="." valign="top">80 (87.0)</td><td align="char" char="." valign="top">77.9&#x2010;92.8</td><td align="char" char="." valign="top">70 (76.1)</td><td align="char" char="." valign="top">65.9&#x2010;84.1</td><td align="char" char="." valign="top">72 (78.3)</td><td align="char" char="." valign="top">68.2&#x2010;85.9</td><td align="char" char="." valign="top">64 (69.6)</td><td align="char" char="." valign="top">59.0&#x2010;78.5</td><td align="char" char="." valign="top">.004</td><td align="char" char="." valign="top">.01</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table3fn7">g</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>c</td><td align="char" char="." valign="top">80 (87.0)</td><td align="char" char="." valign="top">77.9&#x2010;92.8</td><td align="char" char="." valign="top">64 (69.6)</td><td align="char" char="." valign="top">59.0&#x2010;78.5</td><td align="char" char="." valign="top">65 (70.7)</td><td align="char" char="." valign="top">60.1&#x2010;79.5</td><td align="char" char="." valign="top">66 (71.7)</td><td align="char" char="." valign="top">61.2&#x2010;80.4</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table3fn7">g</xref></sup></td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table3fn7">g</xref></sup></td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table3fn7">g</xref></sup></td></tr><tr><td align="left" valign="top">Breast (n=108)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>a</td><td align="char" char="." valign="top">65 (60.2)</td><td align="char" char="." valign="top">50.3&#x2010;69.3</td><td align="char" char="." valign="top">73 (67.6)</td><td align="char" char="." valign="top">57.8&#x2010;76.1</td><td align="char" char="." valign="top">70 (64.8)</td><td align="char" char="." valign="top">55.0&#x2010;73.6</td><td align="char" char="." valign="top">72 (66.7)</td><td align="char" char="." valign="top">56.9&#x2010;75.3</td><td align="char" char="." valign="top">.24</td><td align="char" char="." valign="top">.53</td><td align="char" char="." valign="top">.36</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>b</td><td align="char" char="." valign="top">65 (60.2)</td><td align="char" char="." valign="top">50.3&#x2010;69.3</td><td align="char" char="." valign="top">71 (65.7)</td><td align="char" char="." valign="top">55.9&#x2010;74.4</td><td align="char" char="." valign="top">71 (65.7)</td><td align="char" char="." valign="top">55.9&#x2010;74.4</td><td align="char" char="." valign="top">70 (64.8)</td><td align="char" char="." valign="top">55.0&#x2010;73.6</td><td align="char" char="." valign="top">.42</td><td align="char" char="." valign="top">.43</td><td align="char" char="." valign="top">.56</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>c</td><td align="char" char="." valign="top">65 (60.2)</td><td align="char" char="." valign="top">50.3&#x2010;69.3</td><td align="char" char="." valign="top">71 (65.7)</td><td align="char" char="." valign="top">55.9&#x2010;74.4</td><td align="char" char="." valign="top">70 (64.8)</td><td align="char" char="." valign="top">55.0&#x2010;73.6</td><td align="char" char="." valign="top">69 (63.9)</td><td align="char" char="." valign="top">54.0&#x2010;72.7</td><td align="char" char="." valign="top">.42</td><td align="char" char="." valign="top">.50</td><td align="char" char="." valign="top">.65</td></tr><tr><td align="left" valign="top">BI-RADS<sup><xref ref-type="table-fn" rid="table3fn10">j</xref></sup> (n=108)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>a</td><td align="char" char="." valign="top">108 (100.0)</td><td align="char" char="." valign="top">95.7&#x2010;100.0</td><td align="char" char="." valign="top">63 (58.3)</td><td align="char" char="." valign="top">48.4&#x2010;67.6</td><td align="char" char="." valign="top">60 (55.6)</td><td align="char" char="." valign="top">45.7&#x2010;65.0</td><td align="char" char="." valign="top">65 (60.2)</td><td align="char" char="." valign="top">50.3&#x2010;69.3</td><td align="left" valign="top">NA<sup><xref ref-type="table-fn" rid="table3fn11">k</xref></sup></td><td align="left" valign="top">NA</td><td align="left" valign="top">NA</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>b</td><td align="char" char="." valign="top">108 (100.0)</td><td align="char" char="." valign="top">95.7&#x2010;100.0</td><td align="char" char="." valign="top">66 (61.1)</td><td align="char" char="." valign="top">51.2&#x2010;70.2</td><td align="char" char="." valign="top">54 (50.0)</td><td align="char" char="." valign="top">40.7&#x2010;59.3</td><td align="char" char="." valign="top">69 (63.9)</td><td align="char" char="." valign="top">54.0&#x2010;72.7</td><td align="left" valign="top">NA</td><td align="left" valign="top">NA</td><td align="left" valign="top">NA</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>c</td><td align="char" char="." valign="top">108 (100.0)</td><td align="char" char="." valign="top">95.7&#x2010;100.0</td><td align="char" char="." valign="top">69 (63.9)</td><td align="char" char="." valign="top">54.0&#x2010;72.7</td><td align="char" char="." valign="top">57 (52.8)</td><td align="char" char="." valign="top">43.0&#x2010;62.4</td><td align="char" char="." valign="top">70 (64.8)</td><td align="char" char="." valign="top">55.0&#x2010;73.6</td><td align="left" valign="top">NA</td><td align="left" valign="top">NA</td><td align="left" valign="top">NA</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Comparisons between LLMs and radiologists were made using the McNemar test. </p></fn><fn id="table3fn2"><p><sup>b</sup>DeepSeek: DeepSeek-R1.</p></fn><fn id="table3fn3"><p><sup>c</sup>Gemini: Gemini 2.5 Pro.</p></fn><fn id="table3fn4"><p><sup>d</sup>GPT: GPT-4o.</p></fn><fn id="table3fn5"><p><sup>e</sup><italic>P</italic> values displayed are nominal (unadjusted).</p></fn><fn id="table3fn6"><p><sup>f</sup>Basic patient information and imaging findings.</p></fn><fn id="table3fn7"><p><sup>g</sup>Effects that remained significant after Holm-Bonferroni correction.</p></fn><fn id="table3fn8"><p><sup>h</sup>Scenario A plus chief complaint or clinical history.</p></fn><fn id="table3fn9"><p><sup>i</sup>Scenario B plus key laboratory results.</p></fn><fn id="table3fn10"><p><sup>j</sup>BI-RADS: Breast Imaging Reporting and Data System.</p></fn><fn id="table3fn11"><p><sup>k</sup>NA: not applicable.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Impact of clinical information on diagnostic accuracy. Bar graphs illustrate the diagnostic accuracy across 3 progressive input scenarios in the (A) liver, (B) lung, and (C) breast cohorts. Performance is evaluated for benign-malignant differentiation, disease diagnosis, and Breast Imaging Reporting and Data System (BI-RADS) classification. We defined 3 progressive information input scenarios: A, basic patient information and imaging findings; B, scenario A plus chief complaint or clinical history; and C, scenario B plus key laboratory results. DeepSeek: DeepSeek-R1; Gemini: Gemini 2.5 Pro; GPT: GPT-4o.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e94904_fig02.png"/></fig></sec><sec id="s3-2-2"><title>Liver Cohort</title><p>As shown in <xref ref-type="fig" rid="figure2">Figure 2A</xref>, the diagnostic accuracy in scenario C was higher than in scenario A for DeepSeek-R1 (77/101, 76.2%, vs 69/101, 68.3%), Gemini 2.5 Pro (76/101, 75.2%, vs 69/101, 68.3%), and GPT-4o (73/101, 72.3%, vs 65/101, 64.4%). No significant difference was observed between the scenarios (all adjusted <italic>P</italic>&#x003E;.99).</p></sec><sec id="s3-2-3"><title>Lung Cohort</title><p>As shown in <xref ref-type="fig" rid="figure2">Figure 2B</xref>, the DeepSeek-R1 and Gemini 2.5 Pro models showed numerically higher diagnostic accuracy in scenario B than in scenario A, without reaching statistical significance (DeepSeek-R1: 70/92, 76.1%, vs 63/92, 68.5%; Gemini 2.5 Pro: 72/92, 78.3%, vs 69/92, 75.0%; all adjusted <italic>P</italic>&#x003E;.99). In contrast, GPT-4o achieved its highest numerical accuracy in scenario A (68/92, 73.9%), which was not significantly different from scenario B (64/92, 69.6%) or scenario C (66/92, 71.7%; all adjusted <italic>P</italic>&#x003E;.99).</p></sec><sec id="s3-2-4"><title>Breast Cohort</title><p>As shown in <xref ref-type="fig" rid="figure2">Figure 2C</xref>, the diagnostic accuracy of DeepSeek-R1 was numerically highest in scenario A. It decreased numerically with the addition of laboratory tests (scenario C), although no significant difference was found between scenarios A and C (73/108, 67.6%, vs 71/108, 65.7%; adjusted <italic>P</italic>&#x003E;.99). For GPT-4o, diagnostic accuracy was numerically highest in scenario A (72/108, 66.7%), with no evidence of a significant difference compared with scenario B (70/108, 64.8%) or scenario C (69/108, 63.9%; all adjusted <italic>P</italic>&#x003E;.99).</p></sec></sec><sec id="s3-3"><title>Comparison of the Diagnostic Accuracy in Different Models</title><p>As shown in <xref ref-type="fig" rid="figure3">Figure 3</xref>, the diagnostic performance of the 3 LLMs was compared with that of radiologists within identical input scenarios.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Comparison of diagnostic accuracy among 3 large language models (LLMs). Bar graphs compare the performance of LLMs within the same input scenarios for the (A) liver, (B) lung, and (C) breast cohorts. Metrics include benign-malignant differentiation, disease diagnosis, and Breast Imaging Reporting and Data System (BI-RADS) classification. We defined 3 progressive information input scenarios: A, basic patient information and imaging findings; B, scenario A plus chief complaint or clinical history; and C, scenario B plus key laboratory results. DeepSeek: DeepSeek-R1; Gemini: Gemini 2.5 Pro; GPT: GPT-4o.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e94904_fig03.png"/></fig><sec id="s3-3-1"><title>Liver Cohort</title><p>As shown in <xref ref-type="fig" rid="figure3">Figure 3A</xref>, radiologists achieved the highest diagnostic accuracy for both benign-malignant differentiation (91/101, 90.1%) and disease diagnosis (82/101, 81.2%). Among the 3 models, DeepSeek-R1 demonstrated the highest accuracy. Regarding benign-malignant differentiation, no evidence of a significant difference was found between DeepSeek-R1 (scenario A: 85/101, 84.2%; scenarios B and C: 89/101, 88.1%) and radiologists across all scenarios (all adjusted <italic>P</italic>&#x003E;.99). Regarding disease diagnosis, the accuracy of the DeepSeek-R1 was 68.3% (69/101) in scenario A, 72.3% (73/101) in scenario B, and 76.2% (77/101) in scenario C. No statistically significant difference was found between the DeepSeek-R1 and radiologists in any scenario after Holm-Bonferroni correction, although numerical trends were noted in scenario A (unadjusted <italic>P</italic>=.004; adjusted <italic>P</italic>=.09) and scenario B (unadjusted <italic>P</italic>=.02; adjusted <italic>P</italic>=.35). In scenario C, no evidence of a difference was observed (adjusted <italic>P</italic>&#x003E;.99).</p></sec><sec id="s3-3-2"><title>Lung Cohort</title><p>As shown in <xref ref-type="fig" rid="figure3">Figure 3B</xref>, radiologists achieved the highest diagnostic accuracy for both benign-malignant differentiation and disease diagnosis (80/92, 87.0% for both). In scenarios A and B, the Gemini 2.5 Pro demonstrated the highest accuracy among the 3 models. In the unadjusted analyses, its accuracy was significantly lower than that of the radiologists for both benign-malignant differentiation (scenario A: 70/92, 76.1%; scenario B: 73/92, 79.3%; all unadjusted <italic>P</italic>&#x003C;.05) and disease diagnosis (scenario A: 69/92, 75.0%; scenario B: 72/92, 78.3%; all unadjusted <italic>P</italic>&#x003C;.05); however, these differences did not reach statistical significance after Holm-Bonferroni correction (all adjusted <italic>P</italic>&#x003E;.05). In scenario C, regarding benign-malignant differentiation, the DeepSeek-R1 achieved the highest accuracy among the models (69/92, 75.0%), but no statistically significant difference was found between DeepSeek-R1 and radiologists (adjusted <italic>P</italic>=.20). Regarding disease diagnosis, the GPT-4o achieved the highest accuracy (66/92, 71.7%), which was also significantly lower than that of radiologists (adjusted <italic>P</italic>=.01).</p></sec><sec id="s3-3-3"><title>Breast Cohort</title><p>As shown in <xref ref-type="fig" rid="figure3">Figure 3C</xref>, regarding benign-malignant differentiation, DeepSeek-R1 achieved the highest accuracy among the 3 models across all scenarios (scenarios A and C: 85/108, 78.7%; scenario B: 87/108, 80.6%). In the unadjusted analyses, its accuracy was significantly lower than that of radiologists (99/108, 91.7%) in all scenarios (all unadjusted <italic>P</italic>&#x003C;.05); however, these differences did not reach statistical significance after Holm-Bonferroni correction (all adjusted <italic>P</italic>&#x003E;.05). Regarding disease diagnosis, DeepSeek-R1 also demonstrated the highest accuracy among the models (scenario A: 73/108, 67.6%; scenarios B and C: 71/108, 65.7%). Although the accuracy was numerically higher than that of radiologists (65/108, 60.2%), no evidence of a significant difference was found (all adjusted <italic>P</italic>&#x003E;.99; <xref ref-type="table" rid="table2">Tables 2</xref> and <xref ref-type="table" rid="table3">3</xref>).</p></sec></sec><sec id="s3-4"><title>Comparison of Diagnostic Accuracy in Different Diseases</title><p>Regarding benign-malignant differentiation, the 3 LLMs consistently achieved their highest accuracy in the liver lesions across all scenarios (range 77.2%&#x2010;88.1%; <xref ref-type="table" rid="table2">Table 2</xref> and <xref ref-type="fig" rid="figure4">Figure 4A</xref>). Regarding disease diagnosis, the lowest accuracy was observed in the breast cohort across all models and scenarios (range 63.9%&#x2010;67.6%; <xref ref-type="table" rid="table3">Table 3</xref> and <xref ref-type="fig" rid="figure4">Figure 4B</xref>).</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Bar graphs illustrate accuracy for (A) benign-malignant differentiation and (B) disease diagnosis. Results are shown for the liver, lung, and breast cohorts across 3 input scenarios for each model. We defined 3 progressive information input scenarios: A, basic patient information and imaging findings; B, scenario A plus chief complaint or clinical history; and C, scenario B plus key laboratory results. DeepSeek: DeepSeek-R1; Gemini: Gemini 2.5 Pro; GPT: GPT-4o.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e94904_fig04.png"/></fig></sec><sec id="s3-5"><title>Diagnostic Accuracy of Benign and Malignant Subgroups</title><p>Diagnostic accuracy and 95% CIs for all cases, stratified by benign and malignant status, are summarized in <xref ref-type="table" rid="table4">Table 4</xref>. <xref ref-type="fig" rid="figure5">Figures 5</xref> and <xref ref-type="fig" rid="figure6">6</xref> illustrate model accuracy across these subgroups. Overall, for both benign-malignant differentiation and disease diagnosis, the LLMs demonstrated higher accuracy for malignant lesions (benign-malignant differentiation: range 88.2%&#x2010;100%; and disease diagnosis: range 78.9%&#x2010;100%) than for benign lesions (benign-malignant differentiation: range 4.2%&#x2010;65.6%; disease diagnosis: range 0%&#x2010;53.1%). Notably, in the lung cohort diagnosis, GPT-4o achieved 0% accuracy for benign lesions across all 3 scenarios.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Diagnostic accuracy of large language models stratified by benign and malignant.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Parameter</td><td align="left" valign="top" colspan="3">Benign, % (95% CI)</td><td align="left" valign="top" colspan="3">Malignant, % (95% CI)</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">Liver (n=101)</td><td align="left" valign="top">Lung (n=92)</td><td align="left" valign="top">Breast (n=108)</td><td align="left" valign="top">Liver (n=101)</td><td align="left" valign="top">Lung (n=92)</td><td align="left" valign="top">Breast (n=108)</td></tr></thead><tbody><tr><td align="left" valign="top">Differentiation<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Radiologist</td><td align="char" char="." valign="top">71.9 (53.0&#x2010;85.6)</td><td align="char" char="." valign="top">50.0 (31.4&#x2010;68.6)</td><td align="char" char="." valign="top">87.5 (70.1&#x2010;95.9)</td><td align="char" char="." valign="top">98.6 (91.1&#x2010;99.9)</td><td align="char" char="." valign="top">100.0 (93.3&#x2010;100.0)</td><td align="char" char="." valign="top">93.4 (84.7&#x2010;97.6)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup>-a<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="char" char="." valign="top">56.2 (37.9&#x2010;73.2)</td><td align="char" char="." valign="top">20.8 (7.9&#x2010;42.7)</td><td align="char" char="." valign="top">37.5 (21.7&#x2010;56.3)</td><td align="char" char="." valign="top">97.1 (89.0&#x2010;99.5)</td><td align="char" char="." valign="top">91.2 (81.1&#x2010;96.4)</td><td align="char" char="." valign="top">96.1 (88.1&#x2010;99.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek-b<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="char" char="." valign="top">65.6 (46.8&#x2010;80.8)</td><td align="char" char="." valign="top">25.0 (10.6&#x2010;47.1)</td><td align="char" char="." valign="top">40.6 (24.2&#x2010;59.2)</td><td align="char" char="." valign="top">98.6 (91.1&#x2010;99.9)</td><td align="char" char="." valign="top">95.6 (86.8&#x2010;98.9)</td><td align="char" char="." valign="top">97.4 (90.0&#x2010;99.5)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek-c<sup><xref ref-type="table-fn" rid="table4fn5">e</xref></sup></td><td align="char" char="." valign="top">65.6 (46.8&#x2010;80.8)</td><td align="char" char="." valign="top">37.5 (19.6&#x2010;59.2)</td><td align="char" char="." valign="top">34.4 (19.2&#x2010;53.2)</td><td align="char" char="." valign="top">98.6 (91.1&#x2010;99.9)</td><td align="char" char="." valign="top">88.2 (77.6&#x2010;94.4)</td><td align="char" char="." valign="top">97.4 (90.0&#x2010;99.5)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini<sup><xref ref-type="table-fn" rid="table4fn6">f</xref></sup>-a</td><td align="char" char="." valign="top">46.9 (29.5&#x2010;65.0)</td><td align="char" char="." valign="top">8.3 (1.5&#x2010;28.5)</td><td align="char" char="." valign="top">28.1 (14.4&#x2010;47.0)</td><td align="char" char="." valign="top">100.0 (93.4&#x2010;100.0)</td><td align="char" char="." valign="top">100.0 (93.3&#x2010;100.0)</td><td align="char" char="." valign="top">98.7 (91.9&#x2010;99.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini-b</td><td align="char" char="." valign="top">43.8 (26.8&#x2010;62.1)</td><td align="char" char="." valign="top">20.8 (7.9&#x2010;42.7)</td><td align="char" char="." valign="top">28.1 (14.4&#x2010;47.0)</td><td align="char" char="." valign="top">98.6 (91.1&#x2010;99.9)</td><td align="char" char="." valign="top">100.0 (93.3&#x2010;100.0)</td><td align="char" char="." valign="top">98.7 (91.9&#x2010;99.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini-c</td><td align="char" char="." valign="top">53.1 (35.0&#x2010;70.5)</td><td align="char" char="." valign="top">4.2 (0.2&#x2010;23.1)</td><td align="char" char="." valign="top">28.1 (14.4&#x2010;47.0)</td><td align="char" char="." valign="top">98.6 (91.1&#x2010;99.9)</td><td align="char" char="." valign="top">95.6 (86.8&#x2010;98.9)</td><td align="char" char="." valign="top">98.7 (91.9&#x2010;99.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT<sup><xref ref-type="table-fn" rid="table4fn7">g</xref></sup>-a</td><td align="char" char="." valign="top">28.1 (14.4&#x2010;47.0)</td><td align="char" char="." valign="top">4.2 (0.2&#x2010;23.1)</td><td align="char" char="." valign="top">25.0 (12.1&#x2010;43.8)</td><td align="char" char="." valign="top">100.0 (93.4&#x2010;100.0)</td><td align="char" char="." valign="top">100.0 (93.3&#x2010;100.0)</td><td align="char" char="." valign="top">97.4 (90.0&#x2010;99.5)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-b</td><td align="char" char="." valign="top">40.6 (24.2&#x2010;59.2)</td><td align="char" char="." valign="top">12.5 (3.3&#x2010;33.5)</td><td align="char" char="." valign="top">21.9 (9.9&#x2010;40.4)</td><td align="char" char="." valign="top">100.0 (93.4&#x2010;100.0)</td><td align="char" char="." valign="top">94.1 (84.9&#x2010;98.1)</td><td align="char" char="." valign="top">97.4 (90.0&#x2010;99.5)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-c</td><td align="char" char="." valign="top">56.2 (37.9&#x2010;73.2)</td><td align="char" char="." valign="top">4.2 (0.2&#x2010;23.1)</td><td align="char" char="." valign="top">25.0 (12.1&#x2010;43.8)</td><td align="char" char="." valign="top">98.6 (91.1&#x2010;99.9)</td><td align="char" char="." valign="top">97.1 (88.8&#x2010;99.5)</td><td align="char" char="." valign="top">96.1 (88.1&#x2010;99.0)</td></tr><tr><td align="left" valign="top">Diagnosis<sup><xref ref-type="table-fn" rid="table4fn8">h</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Radiologist</td><td align="char" char="." valign="top">62.5 (43.7&#x2010;78.3)</td><td align="char" char="." valign="top">50.0 (31.4&#x2010;68.6)</td><td align="char" char="." valign="top">68.8 (49.9&#x2010;83.3)</td><td align="char" char="." valign="top">89.9 (79.6&#x2010;95.5)</td><td align="char" char="." valign="top">100.0 (93.3&#x2010;100.0)</td><td align="char" char="." valign="top">56.6 (44.7&#x2010;67.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek-a</td><td align="char" char="." valign="top">43.8 (26.8&#x2010;62.1)</td><td align="char" char="." valign="top">12.5 (3.3&#x2010;33.5)</td><td align="char" char="." valign="top">34.4 (19.2&#x2010;53.2)</td><td align="char" char="." valign="top">79.7 (68.0&#x2010;88.1)</td><td align="char" char="." valign="top">88.2 (77.6&#x2010;94.4)</td><td align="char" char="." valign="top">81.6 (70.7&#x2010;89.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek-b</td><td align="char" char="." valign="top">40.6 (24.2&#x2010;59.2)</td><td align="char" char="." valign="top">20.8 (7.9&#x2010;42.7)</td><td align="char" char="." valign="top">34.4 (19.2&#x2010;53.2)</td><td align="char" char="." valign="top">87.0 (76.2&#x2010;93.5)</td><td align="char" char="." valign="top">95.6 (86.8&#x2010;98.9)</td><td align="char" char="." valign="top">78.9 (67.8&#x2010;87.1)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek-c</td><td align="char" char="." valign="top">53.1 (35.0&#x2010;70.5)</td><td align="char" char="." valign="top">16.7 (5.5&#x2010;38.2)</td><td align="char" char="." valign="top">28.1 (14.4&#x2010;47.0)</td><td align="char" char="." valign="top">87.0 (76.2&#x2010;93.5)</td><td align="char" char="." valign="top">88.2 (77.6&#x2010;94.4)</td><td align="char" char="." valign="top">81.6 (70.7&#x2010;89.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini-a</td><td align="char" char="." valign="top">37.5 (21.7&#x2010;56.3)</td><td align="char" char="." valign="top">4.2 (0.2&#x2010;23.1)</td><td align="char" char="." valign="top">21.9 (9.9&#x2010;40.4)</td><td align="char" char="." valign="top">82.6 (71.2&#x2010;90.3)</td><td align="char" char="." valign="top">100.0 (93.3&#x2010;100.0)</td><td align="char" char="." valign="top">82.9 (72.2&#x2010;90.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini-b</td><td align="char" char="." valign="top">37.5 (21.7&#x2010;56.3)</td><td align="char" char="." valign="top">16.7 (5.5&#x2010;38.2)</td><td align="char" char="." valign="top">21.9 (9.9&#x2010;40.4)</td><td align="char" char="." valign="top">87.0 (76.2&#x2010;93.5)</td><td align="char" char="." valign="top">100.0 (93.3&#x2010;100.0)</td><td align="char" char="." valign="top">84.2 (73.6&#x2010;91.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini-c</td><td align="char" char="." valign="top">46.9 (29.5&#x2010;65.0)</td><td align="char" char="." valign="top">4.2 (0.2&#x2010;23.1)</td><td align="char" char="." valign="top">25.0 (12.1&#x2010;43.8)</td><td align="char" char="." valign="top">88.4 (77.9&#x2010;94.5)</td><td align="char" char="." valign="top">94.1 (84.9&#x2010;98.1)</td><td align="char" char="." valign="top">81.6 (70.7&#x2010;89.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-a</td><td align="char" char="." valign="top">15.6 (5.9&#x2010;33.5)</td><td align="char" char="." valign="top">0.0 (0.0&#x2010;17.2)</td><td align="char" char="." valign="top">25.0 (12.1&#x2010;43.8)</td><td align="char" char="." valign="top">87.0 (76.2&#x2010;93.5)</td><td align="char" char="." valign="top">100.0 (93.3&#x2010;100.0)</td><td align="char" char="." valign="top">84.2 (73.6&#x2010;91.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-b</td><td align="char" char="." valign="top">25.0 (12.1&#x2010;43.8)</td><td align="char" char="." valign="top">0.0 (0.0&#x2010;17.2)</td><td align="char" char="." valign="top">18.8 (7.9&#x2010;37.0)</td><td align="char" char="." valign="top">89.9 (79.6&#x2010;95.5)</td><td align="char" char="." valign="top">94.1 (84.9&#x2010;98.1)</td><td align="char" char="." valign="top">84.2 (73.6&#x2010;91.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-c</td><td align="char" char="." valign="top">31.2 (16.7&#x2010;50.1)</td><td align="char" char="." valign="top">0.0 (0.0&#x2010;17.2)</td><td align="char" char="." valign="top">18.8 (7.9&#x2010;37.0)</td><td align="char" char="." valign="top">91.3 (71.4&#x2010;96.4)</td><td align="char" char="." valign="top">97.1 (88.8&#x2010;99.5)</td><td align="char" char="." valign="top">82.9 (72.2&#x2010;90.2)</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Benign-malignant differentiation.</p></fn><fn id="table4fn2"><p><sup>b</sup>DeepSeek: DeepSeek-R1.</p></fn><fn id="table4fn3"><p><sup>c</sup>Basic patient information and imaging findings.</p></fn><fn id="table4fn4"><p><sup>d</sup>Scenario A plus chief complaint or clinical history.</p></fn><fn id="table4fn5"><p><sup>e</sup>Scenario B plus key laboratory results.</p></fn><fn id="table4fn6"><p><sup>f</sup>Gemini: Gemini 2.5 Pro.</p></fn><fn id="table4fn7"><p><sup>g</sup>GPT: GPT-4o.</p></fn><fn id="table4fn8"><p><sup>h</sup>Disease diagnosis.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Subgroup analysis for benign-malignant differentiation. Diagnostic accuracy of 3 large language models is stratified by benign and malignant lesion status in the (A) liver, (B) lung, and (C) breast cohorts. We defined 3 progressive information input scenarios: A, basic patient information and imaging findings; B, scenario A plus chief complaint or clinical history; and C, scenario B plus key laboratory results. DeepSeek: DeepSeek-R1; Gemini: Gemini 2.5 Pro; GPT: GPT-4o.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e94904_fig05.png"/></fig><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Subgroup analysis for disease diagnosis. Diagnostic accuracy of 3 large language models is stratified by benign and malignant lesion status in the (A) liver, (B) lung, and (C) breast cohorts. We defined 3 progressive information input scenarios: A, basic patient information and imaging findings; B, scenario A plus chief complaint or clinical history; and C, scenario B plus key laboratory results. DeepSeek: DeepSeek-R1; Gemini: Gemini 2.5 Pro; GPT: GPT-4o.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e94904_fig06.png"/></fig></sec><sec id="s3-6"><title>BI-RADS Classification Consistency Assessment</title><p>Overall, diagnostic accuracy was highest for GPT-4o (range 60.2%&#x2010;64.8%), followed by DeepSeek-R1 (range 58.3%&#x2010;63.9%) and Gemini 2.5 Pro (range 50%&#x2010;55.6%; <xref ref-type="table" rid="table3">Table 3</xref>, <xref ref-type="fig" rid="figure2">Figure 2</xref>C and <xref ref-type="fig" rid="figure3">Figure 3</xref>C). A complete matrix of <italic>P</italic> values for all pairwise comparisons regarding BI-RADS classifications is provided in Table S3 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>While numerous studies have investigated the utility of LLMs for generating the impression section of radiology reports from images or findings [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref22">22</xref>], the diagnostic accuracy of these models across diverse clinical scenarios remains underexplored. This study evaluated the impact of varying clinical input information on the diagnostic accuracy of general-purpose LLMs (DeepSeek-R1, Gemini 2.5 Pro, and GPT-4o) and compared their performance with that of radiologists in real-world diagnostic tasks. While diagnostic accuracy generally improved with the addition of clinical information, this improvement was heterogeneous across models and disease types. In most cases, the addition of clinical history and laboratory results improved the diagnostic accuracy of the model (eg, in the liver cohort, the diagnostic accuracy in scenario C was higher than in scenario A for DeepSeek-R1 [77/101, 76.2%, vs 69/101, 68.3%; unadjusted <italic>P</italic>=.06, adjusted <italic>P</italic>&#x003E;.99], Gemini 2.5 Pro [76/101, 75.2%, vs 69/101, 68.3%; unadjusted <italic>P</italic>=.10; adjusted <italic>P</italic>&#x003E;.99], and GPT-4o [73/101, 72.3%, vs 65/101, 64.4%; unadjusted <italic>P</italic>=.06; adjusted <italic>P</italic>&#x003E;.99]), and this result is consistent with those of Ueda et al [<xref ref-type="bibr" rid="ref24">24</xref>]. However, we also observed exceptions: In the lung cohort, GPT-4o achieved its highest accuracy in scenario A (68/92, 73.9%). In the breast cohort, the diagnostic accuracy of DeepSeek-R1 peaked in scenario A (73/108, 67.6%) but decreased to 65.7% (71/108) with the addition of laboratory tests; there was no evidence of a difference between the scenarios (adjusted <italic>P</italic>&#x003E;.99). This suggests that the addition of laboratory test results has less or even a negative impact on lung and breast diseases in diagnosis. This discrepancy may be because serum tumor markers such as alpha-fetoprotein, carbohydrate antigen 19&#x2010;9, and carcinoembryonic antigen play an essential role in the diagnosis of liver diseases [<xref ref-type="bibr" rid="ref31">31</xref>]. It is also possible that the LLM may misinterpret the addition of extra information as &#x201C;noise&#x201D; rather than &#x201C;signal,&#x201D; thereby interfering with its diagnostic logic based on image findings and history. However, we emphasize that this is only a descriptive analogy, not a mechanistic claim, and the precise reasons for this performance degradation remain unclear.</p></sec><sec id="s4-2"><title>Comparison With Prior Work</title><p>Although providing comprehensive clinical information was associated with numerically higher diagnostic accuracy, the improvement was not statistically significant, and overall model performance remained inferior to that of radiologists. Zhang et al [<xref ref-type="bibr" rid="ref20">20</xref>] reported similar results, noting that even their specialized model struggled to match radiologists&#x2019; performance in key dimensions, such as providing a &#x201C;specific diagnosis&#x201D; (647/1014, 63.8%). Interestingly, we observed that specific models achieved higher diagnostic accuracy in the breast cohort than radiologists. This discrepancy likely stems from the study&#x2019;s rigorous evaluation criteria, which classified reports with unspecified case names as BI-RADS categories 0&#x2013;4b as &#x201C;inconsistent,&#x201D; a rule that may have artificially underestimated the radiologists&#x2019; performance.</p><p>Kim et al [<xref ref-type="bibr" rid="ref32">32</xref>] reported performance variations across radiologic subspecialties, a finding consistent with our results. Additionally, while the model demonstrated good diagnostic accuracy for common diseases, such as hepatocellular carcinoma and lung adenocarcinoma, a granular analysis of the &#x201C;other&#x201D; category revealed poor performance when handling rare and complex lesions. The model exhibited a tendency toward &#x201C;shortcut learning,&#x201D; a phenomenon frequently driven by confounding bias [<xref ref-type="bibr" rid="ref33">33</xref>]. For example, Rueckel et al [<xref ref-type="bibr" rid="ref34">34</xref>] demonstrated that an AI system designed to detect pneumothorax relied on the presence of thoracic drainage tubes instead of learning features intrinsic to the pathology itself. Similarly, in our study, when hepatic adenoma was misclassified as focal nodular hyperplasia (FNH), the model appeared to treat arterial phase hyperenhancement and persistent enhancement as sufficient criteria for FNH, thereby failing to incorporate a key discriminative feature where the central scar in FNH typically demonstrates delayed-phase hyperintensity. More broadly, the model appeared to rely heavily on combinations of demographic features and coarse imaging patterns, which may bias predictions toward population-level associations rather than individualized assessments. For instance, cases characterized as &#x201C;young female with persistent enhancement&#x201D; were frequently predicted as hepatic adenoma, whereas &#x201C;young male without underlying liver disease but with a central scar&#x201D; tended to be classified as FNH. A similar pattern was observed in pulmonary diagnoses, where a strong association between &#x201C;middle-aged female&#x201D; and lung adenocarcinoma appeared to underpin numerous misclassifications. Although lung adenocarcinoma is indeed more prevalent among women who have never smoked [<xref ref-type="bibr" rid="ref35">35</xref>], the model&#x2019;s reliance on statistical regularities may bias predictions toward more common diseases. For example, lung adenocarcinoma and sclerosing pneumocytoma share similar demographic profiles because both are more frequent in middle-aged women. In such cases, dependence on population-level patterns may lead the model to favor the more prevalent diagnosis, resulting in the underrecognition of rarer conditions such as sclerosing pneumocytoma. This limitation is particularly critical in clinical settings where the tolerance for diagnostic error is low [<xref ref-type="bibr" rid="ref36">36</xref>]. We observed that GPT-4o achieved 0% accuracy for benign lung lesions across all 3 scenarios, indicating that every benign lesion in the lung cohort was classified as malignant. The clinical safety implications of this finding warrant particular attention. In a real-world clinical setting, such systematic overdiagnosis could lead to unnecessary invasive procedures, heightened patient anxiety, and substantial downstream health care costs. This failure mode underscores a broader deployment risk: current general-purpose LLMs may lack the conservative judgment required for low-tolerance clinical tasks, particularly when distinguishing benign from malignant conditions. In addition, automation bias in the model may lead to an imbalanced weighting of multimodal evidence. During diagnostic reasoning, the model may assign disproportionate importance to low-specificity imaging features, such as &#x201C;possible biliary dilatation,&#x201D; while failing to adequately integrate critical negative clinical findings and laboratory results. Consistent with our observations, Zack et al [<xref ref-type="bibr" rid="ref37">37</xref>] reported that aggregate evaluation metrics can obscure underlying biases in individual cases, particularly those relating to demographic attributes.</p><p>We also observed performance variations among the models. Results indicated that model performance was consistent across input scenarios for the liver and breast cohorts. Conversely, performance for the lung cohort exhibited greater variability. We hypothesize that this discrepancy stems from differences in imaging modalities, as MRI is a multiparametric and multisequence modality that provides richer information than CT. Zheng et al [<xref ref-type="bibr" rid="ref21">21</xref>] supported this observation and reported that the odds ratio of MRI versus CT was 3.46, with a 95% CI of 2.56 to 4.67. DeepSeek-R1 demonstrated the highest performance in assessing liver and breast diseases. This finding may be attributable to the use of Chinese-language inputs, given that DeepSeek-R1 was optimized for Chinese text during training, whereas Gemini 2.5 Pro and GPT-4o were not. Prior studies have reported substantial performance variability across languages in general-purpose LLMs. For example, Cozzi et al [<xref ref-type="bibr" rid="ref38">38</xref>] demonstrated that human-LLM agreement for BI-RADS classification decreased markedly when moving from English to Italian and Dutch, while Meddeb et al [<xref ref-type="bibr" rid="ref39">39</xref>] showed that translation quality varies significantly across different model-language pairs.</p></sec><sec id="s4-3"><title>Limitations</title><p>This study has several limitations.</p><p>First, a core limitation of this study is the inherent asymmetry in the comparison between LLMs and human radiologists. The experimental design comprised 3 progressively enriched, text-based clinical scenarios specifically constructed for LLM evaluation. In contrast, the historical radiologist impressions used as a comparator were generated in real-world clinical settings, where radiologists likely had access to comprehensive electronic health records at the time of diagnosis, including prior imaging studies, longitudinal clinical documentation, and laboratory data. This imbalance in information availability introduces a significant source of confounding and limits the validity of direct performance comparisons. This study design, while pragmatic for exploring LLM-based diagnostic reasoning, does not support a controlled or fully comparable assessment of radiologist-level performance. Future prospective studies should adopt a symmetrical evaluation framework, in which both LLMs and radiologists are provided with identical inputs under matched conditions. Such designs would enable a more rigorous and unbiased assessment of comparative diagnostic performance. For breast BI-RADS classification, the final radiologist impressions served as the reference standard. This design makes the model versus radiologist comparison structurally circular, as the radiologist defines the correct answer and necessarily achieves 100% accuracy. Model performance for this task should therefore be interpreted as agreement with the radiologist reference standard rather than as a direct comparison of diagnostic capability. Additionally, in the breast cohort, the strict classification rule used to define diagnostic consistency may have artificially lowered radiologist accuracy, thereby lowering the baseline for comparison with the models. As a result, the observed performance gap in this cohort may partly reflect the stringency of the evaluation framework rather than a true difference in diagnostic capability, representing an inherent comparability constraint. This limitation should therefore be considered when interpreting the comparative findings for the breast cohort.</p><p>Second, an inherent limitation of this study is that the models evaluated text-based descriptive findings rather than the medical images themselves. Consequently, the diagnostic performance ceiling of the models is entirely dependent on the visual perception and descriptive accuracy of the radiologists who originally authored the reports. If a subtle but diagnostically relevant imaging feature was omitted, understated, or ambiguously phrased in the text, the model necessarily lacked access to that information and could not incorporate it into its diagnostic reasoning. Future studies should develop multimodal model architectures that integrate imaging and text inputs to overcome the inherent constraints of single-modality text-based evaluation.</p><p>Third, a limitation concerns the asymmetry in clinical information provided across disease cohorts. In scenario B, breast cases were accompanied by relatively comprehensive clinical histories, including medical, menstrual, and family information. In contrast, liver and lung cases were provided with only the chief complaint. This asymmetry was not due to an inability to retrieve available clinical history, but rather to reflect the absence of equivalently structured data for the liver and lung cohorts in the retrospective data source. This imbalance in contextual information introduces a potential source of confounding and limits the validity of direct cross-organ performance comparisons. While this design partially reflects real-world variability in clinical documentation across organ systems, it nonetheless constitutes a structural inconsistency within the experimental framework. Future prospective studies should adopt standardized clinical information templates across all disease types, ensuring that models are evaluated under matched informational conditions to enable more rigorous and unbiased cross-organ comparisons. Additionally, several aspects of the data input design may have elevated input quality relative to routine clinical practice, thereby limiting external validity. First, in scenario C, we provided only key laboratory markers relevant to each disease type and manually removed the vast majority of laboratory data typically found in real-world electronic health records, effectively and artificially &#x201C;sanitizing&#x201D; the dataset. Second, these laboratory values were supplied without accompanying institutional normal reference ranges; because reference intervals vary by assay and equipment, the models may have been forced to rely on generalized reference intervals from pretraining data, potentially introducing uncertainty into the interpretation of laboratory values and contributing to diagnostic error. Furthermore, the exclusion of incomplete or ambiguous radiology reports, for example, 317 cases in the lung cohort, further elevated input quality relative to routine clinical practice. Future studies should use complete, unfiltered laboratory datasets with explicit institutional reference ranges to more accurately assess model robustness under clinically realistic conditions.</p><p>Fourth, language represents an unavoidable source of potential bias. As the clinical data were provided in Chinese and different models may exhibit varying degrees of native multilingual optimization, the observed performance differences may be partially driven by differences in linguistic comprehension rather than solely reflecting variations in clinical reasoning ability. Future studies incorporating multilingual inputs, as well as models with different primary language optimization strategies, are warranted to more rigorously disentangle linguistic effects from true diagnostic reasoning performance.</p><p>Fifth, the study cohort included only surgically resected cases with histopathological confirmation, which introduces a substantial selection bias. In routine clinical practice, many typical benign lesions (eg, hepatic hemangiomas or breast fibroadenomas) are confidently diagnosed based on imaging alone and therefore do not undergo surgical resection. Consequently, the benign cases in this study are likely enriched for atypical or diagnostically challenging lesions, which may partly explain the relatively low diagnostic accuracy observed for benign diseases. In addition, the statistical power for analyses of benign subgroups was limited, as reflected by wide CIs and unstable estimates. These findings should therefore be interpreted with caution, and the diagnostic performance of the models for benign conditions requires further validation in larger and more representative cohorts.</p><p>Sixth, no prospective sample size calculation was performed prior to data collection. According to the STARD (Standards for Reporting of Diagnostic Accuracy Studies) 2015 guidelines (item 14), a priori sample size estimation is recommended to ensure adequate precision and reliability of diagnostic accuracy studies. The absence of such a formal calculation in this study may have affected the precision of the reported performance estimates. Future prospective studies should incorporate predefined sample size calculations to ensure sufficient statistical power and more precise estimation of diagnostic accuracy.</p><p>Seventh, it should be noted that the evaluation in this study was based on a strict binary rubric that credited only the definitive final diagnosis. The models were prompted to provide both a final diagnosis and a differential diagnosis; however, the differential diagnosis sections were typically very broad and nondiscriminatory. Crediting a case as correct based solely on the presence of the correct diagnosis anywhere in the output would have artificially inflated the apparent performance. We therefore adopted a conservative approach, while acknowledging that the strict rubric may underestimate the models&#x2019; capacity for nuanced clinical reasoning when the correct diagnosis is appropriately prioritized.</p><p>Eighth, all clinical text was manually deidentified prior to input, with patient names, medical record numbers, and institutional identifiers removed. However, certain dates embedded in the clinical history, such as the last menstrual period date and other dated clinical events, were not uniformly removed or generalized during the initial data preparation. This represents a limitation in our deidentification protocol and should be acknowledged as a privacy consideration. Future studies using public LLM interfaces should adopt more rigorous deidentification procedures, including the removal or generalization of all date information, to fully comply with best practices for patient data protection.</p><p>Finally, the study relied on single-shot prompting, which does not fully capture the probabilistic nature and inherent stochasticity of general-purpose LLMs. As a result, the stability, reproducibility, and intramodel variability of diagnostic outputs could not be systematically evaluated, and the potential variation in reasoning across repeated inference runs remains unquantified. In addition, the prompt templates used in this study were relatively basic and have not undergone extensive optimization. As such, the potential benefits of more advanced prompt engineering strategies were not fully explored in this foundational evaluation. To more rigorously assess diagnostic robustness, future studies should adopt multi-iteration inference protocols. Such designs would enable estimation of output variance, systematic evaluation of hallucination rates, and construction of CIs for model-level performance. Coupling repeated inference with refined prompt optimization strategies will be essential for more comprehensively characterizing the intrinsic reliability of these models in clinical applications.</p></sec><sec id="s4-4"><title>Conclusions</title><p>In conclusion, we evaluated the diagnostic accuracy of DeepSeek-R1, Gemini 2.5 Pro, and GPT-4o in liver, lung, and breast diseases across progressive information input scenarios and compared their accuracy with that of radiologists. A numerical trend toward improvement was observed when systematically increasing the dimensionality of clinical information, but no statistically significant difference was demonstrated. Previous studies have demonstrated the potential of LLMs to automatically extract clinical histories [<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref41">41</xref>]. Future clinical AI systems may leverage LLMs to integrate imaging, clinical history, and laboratory data, thereby further improving diagnostic accuracy.</p></sec></sec></body><back><ack><p>We sincerely thank the teams behind ChatGPT, Gemini, and DeepSeek for their contributions to AI development and public access. The authors used Google Gemini for English language polishing and grammar correction during manuscript preparation. All AI-generated outputs were reviewed and revised by the authors, who assume full responsibility for the final content of this manuscript.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the National Natural Science Foundation of China (grant 82102029); National High Level Hospital Clinical Research Funding, China (grant LC2024A09); and Teaching Research Fund of Cancer Hospital of Chinese Academy of Medical Sciences (grant E2024015).</p></sec><sec><title>Data Availability</title><p>The datasets used and analyzed during this study are available from the corresponding author on reasonable request.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: JZ, XW, ZZ</p><p>Data curation: JZ, XW, YZ</p><p>Formal analysis: JZ, YZ</p><p>Funding acquisition: ZZ</p><p>Investigation: JZ, XW, XC</p><p>Methodology: JZ, XZ, ZZ</p><p>Project administration: ZZ</p><p>Supervision: XZ, HZ, ZZ</p><p>Validation: JZ, LZ</p><p>Writing &#x2013; original draft: JZ, ZZ</p><p>Writing &#x2013; review &#x0026; editing: JZ, YZ, ZZ</p><p>YZ is the co-corresponding author and can be reached via email at zyf24@sina.com or phone at 86 13021105004.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">BI-RADS</term><def><p>Breast Imaging Reporting and Data System</p></def></def-item><def-item><term id="abb2">CT</term><def><p>computed tomography</p></def></def-item><def-item><term id="abb3">FNH</term><def><p>focal nodular hyperplasia</p></def></def-item><def-item><term id="abb4">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb5">MRI</term><def><p>magnetic resonance imaging</p></def></def-item><def-item><term id="abb6">STARD</term><def><p>Standards for Reporting of Diagnostic Accuracy Studies</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bhayana</surname><given-names>R</given-names> </name><name name-style="western"><surname>Krishna</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bleakney</surname><given-names>RR</given-names> </name></person-group><article-title>Performance of ChatGPT on a radiology board-style examination: insights into current strengths and limitations</article-title><source>Radiology</source><year>2023</year><month>06</month><volume>307</volume><issue>5</issue><fpage>e230582</fpage><pub-id pub-id-type="doi">10.1148/radiol.230582</pub-id><pub-id pub-id-type="medline">37191485</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>He</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Niu</surname><given-names>W</given-names> </name><etal/></person-group><article-title>From algorithms to operating room: can large language models master China&#x2019;s attending anesthesiology exam? A cross-sectional evaluation</article-title><source>Int J Surg</source><year>2026</year><month>01</month><day>1</day><volume>112</volume><issue>1</issue><fpage>190</fpage><lpage>201</lpage><pub-id pub-id-type="doi">10.1097/JS9.0000000000003406</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Longwell</surname><given-names>JB</given-names> </name><name name-style="western"><surname>Hirsch</surname><given-names>I</given-names> </name><name name-style="western"><surname>Binder</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Performance of large language models on medical oncology examination questions</article-title><source>JAMA Netw Open</source><year>2024</year><month>06</month><day>3</day><volume>7</volume><issue>6</issue><fpage>e2417641</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.17641</pub-id><pub-id pub-id-type="medline">38888919</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Amin</surname><given-names>KS</given-names> </name><name name-style="western"><surname>Davis</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Doshi</surname><given-names>R</given-names> </name><name name-style="western"><surname>Haims</surname><given-names>AH</given-names> </name><name name-style="western"><surname>Khosla</surname><given-names>P</given-names> </name><name name-style="western"><surname>Forman</surname><given-names>HP</given-names> </name></person-group><article-title>Accuracy of ChatGPT, Google Bard, and Microsoft Bing for simplifying radiology reports</article-title><source>Radiology</source><year>2023</year><month>11</month><volume>309</volume><issue>2</issue><fpage>e232561</fpage><pub-id pub-id-type="doi">10.1148/radiol.232561</pub-id><pub-id pub-id-type="medline">37987662</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Salam</surname><given-names>B</given-names> </name><name name-style="western"><surname>Kravchenko</surname><given-names>D</given-names> </name><name name-style="western"><surname>Nowak</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Generative Pre-trained Transformer 4 makes cardiovascular magnetic resonance reports easy to understand</article-title><source>J Cardiovasc Magn Reson</source><year>2024</year><volume>26</volume><issue>1</issue><fpage>101035</fpage><pub-id pub-id-type="doi">10.1016/j.jocmr.2024.101035</pub-id><pub-id pub-id-type="medline">38460841</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lyu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zapadka</surname><given-names>ME</given-names> </name><etal/></person-group><article-title>Translating radiology reports into plain language using ChatGPT and GPT-4 with prompt learning: results, limitations, and potential</article-title><source>Vis Comput Ind Biomed Art</source><year>2023</year><month>05</month><day>18</day><volume>6</volume><issue>1</issue><fpage>9</fpage><pub-id pub-id-type="doi">10.1186/s42492-023-00136-5</pub-id><pub-id pub-id-type="medline">37198498</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rahsepar</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Tavakoli</surname><given-names>N</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>GHJ</given-names> </name><name name-style="western"><surname>Hassani</surname><given-names>C</given-names> </name><name name-style="western"><surname>Abtin</surname><given-names>F</given-names> </name><name name-style="western"><surname>Bedayat</surname><given-names>A</given-names> </name></person-group><article-title>How AI responds to common lung cancer questions: ChatGPT vs Google Bard</article-title><source>Radiology</source><year>2023</year><month>06</month><volume>307</volume><issue>5</issue><fpage>e230922</fpage><pub-id pub-id-type="doi">10.1148/radiol.230922</pub-id><pub-id pub-id-type="medline">37310252</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Haver</surname><given-names>HL</given-names> </name><name name-style="western"><surname>Ambinder</surname><given-names>EB</given-names> </name><name name-style="western"><surname>Bahl</surname><given-names>M</given-names> </name><name name-style="western"><surname>Oluyemi</surname><given-names>ET</given-names> </name><name name-style="western"><surname>Jeudy</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yi</surname><given-names>PH</given-names> </name></person-group><article-title>Appropriateness of breast cancer prevention and screening recommendations provided by ChatGPT</article-title><source>Radiology</source><year>2023</year><month>05</month><volume>307</volume><issue>4</issue><fpage>e230424</fpage><pub-id pub-id-type="doi">10.1148/radiol.230424</pub-id><pub-id pub-id-type="medline">37014239</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Haver</surname><given-names>HL</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>CT</given-names> </name><name name-style="western"><surname>Sirajuddin</surname><given-names>A</given-names> </name><name name-style="western"><surname>Yi</surname><given-names>PH</given-names> </name><name name-style="western"><surname>Jeudy</surname><given-names>J</given-names> </name></person-group><article-title>Use of ChatGPT, GPT-4, and Bard to improve readability of ChatGPT&#x2019;s answers to common questions about lung cancer and lung cancer screening</article-title><source>AJR Am J Roentgenol</source><year>2023</year><month>11</month><volume>221</volume><issue>5</issue><fpage>701</fpage><lpage>704</lpage><pub-id pub-id-type="doi">10.2214/AJR.23.29622</pub-id><pub-id pub-id-type="medline">37341179</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Haver</surname><given-names>HL</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>CT</given-names> </name><name name-style="western"><surname>Sirajuddin</surname><given-names>A</given-names> </name><name name-style="western"><surname>Yi</surname><given-names>PH</given-names> </name><name name-style="western"><surname>Jeudy</surname><given-names>J</given-names> </name></person-group><article-title>Evaluating ChatGPT&#x2019;s accuracy in lung cancer prevention and screening recommendations</article-title><source>Radiol Cardiothorac Imaging</source><year>2023</year><month>08</month><volume>5</volume><issue>4</issue><fpage>e230115</fpage><pub-id pub-id-type="doi">10.1148/ryct.230115</pub-id><pub-id pub-id-type="medline">37693201</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Le Guellec</surname><given-names>B</given-names> </name><name name-style="western"><surname>Lef&#x00E8;vre</surname><given-names>A</given-names> </name><name name-style="western"><surname>Geay</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Performance of an open-source large language model in extracting information from free-text radiology reports</article-title><source>Radiol Artif Intell</source><year>2024</year><month>07</month><volume>6</volume><issue>4</issue><fpage>e230364</fpage><pub-id pub-id-type="doi">10.1148/ryai.230364</pub-id><pub-id pub-id-type="medline">38717292</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lehnen</surname><given-names>NC</given-names> </name><name name-style="western"><surname>Dorn</surname><given-names>F</given-names> </name><name name-style="western"><surname>Wiest</surname><given-names>IC</given-names> </name><etal/></person-group><article-title>Data extraction from free-text reports on mechanical thrombectomy in acute ischemic stroke using ChatGPT: a retrospective analysis</article-title><source>Radiology</source><year>2024</year><month>04</month><volume>311</volume><issue>1</issue><fpage>e232741</fpage><pub-id pub-id-type="doi">10.1148/radiol.232741</pub-id><pub-id pub-id-type="medline">38625006</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fink</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Bischoff</surname><given-names>A</given-names> </name><name name-style="western"><surname>Fink</surname><given-names>CA</given-names> </name><etal/></person-group><article-title>Potential of ChatGPT and GPT-4 for data mining of free-text CT reports on lung cancer</article-title><source>Radiology</source><year>2023</year><month>09</month><volume>308</volume><issue>3</issue><fpage>e231362</fpage><pub-id pub-id-type="doi">10.1148/radiol.231362</pub-id><pub-id pub-id-type="medline">37724963</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bhayana</surname><given-names>R</given-names> </name><name name-style="western"><surname>Nanda</surname><given-names>B</given-names> </name><name name-style="western"><surname>Dehkharghanian</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Large language models for automated synoptic reports and resectability categorization in pancreatic cancer</article-title><source>Radiology</source><year>2024</year><month>06</month><volume>311</volume><issue>3</issue><fpage>e233117</fpage><pub-id pub-id-type="doi">10.1148/radiol.233117</pub-id><pub-id pub-id-type="medline">38888478</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>KW</given-names> </name><name name-style="western"><surname>Lacson</surname><given-names>R</given-names> </name><name name-style="western"><surname>Guenette</surname><given-names>JP</given-names> </name><etal/></person-group><article-title>Use of ChatGPT large language models to extract details of recommendations for additional imaging from free-text impressions of radiology reports</article-title><source>AJR Am J Roentgenol</source><year>2025</year><month>04</month><volume>224</volume><issue>4</issue><fpage>e2432341</fpage><pub-id pub-id-type="doi">10.2214/AJR.24.32341</pub-id><pub-id pub-id-type="medline">39878409</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Ong</surname><given-names>H</given-names> </name><name name-style="western"><surname>Kennedy</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Evaluating GPT-V4 (GPT-4 with vision) on detection of radiologic findings on chest radiographs</article-title><source>Radiology</source><year>2024</year><month>05</month><volume>311</volume><issue>2</issue><fpage>e233270</fpage><pub-id pub-id-type="doi">10.1148/radiol.233270</pub-id><pub-id pub-id-type="medline">38713028</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gertz</surname><given-names>RJ</given-names> </name><name name-style="western"><surname>Dratsch</surname><given-names>T</given-names> </name><name name-style="western"><surname>Bunck</surname><given-names>AC</given-names> </name><etal/></person-group><article-title>Potential of GPT-4 for detecting errors in radiology reports: implications for reporting accuracy</article-title><source>Radiology</source><year>2024</year><month>04</month><volume>311</volume><issue>1</issue><fpage>e232714</fpage><pub-id pub-id-type="doi">10.1148/radiol.232714</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>D</given-names> </name><name name-style="western"><surname>Shin</surname><given-names>HJ</given-names> </name><etal/></person-group><article-title>Large-scale validation of the feasibility of GPT-4 as a proofreading tool for head CT reports</article-title><source>Radiology</source><year>2025</year><month>01</month><volume>314</volume><issue>1</issue><fpage>e240701</fpage><pub-id pub-id-type="doi">10.1148/radiol.240701</pub-id><pub-id pub-id-type="medline">39873601</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sun</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Ong</surname><given-names>H</given-names> </name><name name-style="western"><surname>Kennedy</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Evaluating GPT4 on impressions generation in radiology reports</article-title><source>Radiology</source><year>2023</year><month>06</month><volume>307</volume><issue>5</issue><fpage>e231259</fpage><pub-id pub-id-type="doi">10.1148/radiol.231259</pub-id><pub-id pub-id-type="medline">37367439</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>M</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Constructing a large language model to generate impressions from findings in radiology reports</article-title><source>Radiology</source><year>2024</year><month>09</month><volume>312</volume><issue>3</issue><fpage>e240885</fpage><pub-id pub-id-type="doi">10.1148/radiol.240885</pub-id><pub-id pub-id-type="medline">39287525</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zheng</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>N</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Comparison of a specialized large language model with GPT-4o for CT and MRI radiology report summarization</article-title><source>Radiology</source><year>2025</year><month>08</month><volume>316</volume><issue>2</issue><fpage>e243774</fpage><pub-id pub-id-type="doi">10.1148/radiol.243774</pub-id><pub-id pub-id-type="medline">40892451</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kottlors</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bratke</surname><given-names>G</given-names> </name><name name-style="western"><surname>Rauen</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Feasibility of differential diagnosis based on imaging patterns using a large language model</article-title><source>Radiology</source><year>2023</year><month>07</month><volume>308</volume><issue>1</issue><fpage>e231167</fpage><pub-id pub-id-type="doi">10.1148/radiol.231167</pub-id><pub-id pub-id-type="medline">37404149</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mukherjee</surname><given-names>P</given-names> </name><name name-style="western"><surname>Hou</surname><given-names>B</given-names> </name><name name-style="western"><surname>Suri</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Evaluation of GPT large language model performance on RSNA 2023 case of the day questions</article-title><source>Radiology</source><year>2024</year><month>10</month><volume>313</volume><issue>1</issue><fpage>e240609</fpage><pub-id pub-id-type="doi">10.1148/radiol.240609</pub-id><pub-id pub-id-type="medline">39352277</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ueda</surname><given-names>D</given-names> </name><name name-style="western"><surname>Mitsuyama</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Takita</surname><given-names>H</given-names> </name><etal/></person-group><article-title>ChatGPT&#x2019;s diagnostic performance from patient history and imaging findings on the diagnosis please quizzes</article-title><source>Radiology</source><year>2023</year><month>07</month><volume>308</volume><issue>1</issue><fpage>e231040</fpage><pub-id pub-id-type="doi">10.1148/radiol.231040</pub-id><pub-id pub-id-type="medline">37462501</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sonoda</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Kurokawa</surname><given-names>R</given-names> </name><name name-style="western"><surname>Hagiwara</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Structured clinical reasoning prompt enhances LLM&#x2019;s diagnostic capabilities in diagnosis please quiz cases</article-title><source>Jpn J Radiol</source><year>2025</year><month>04</month><volume>43</volume><issue>4</issue><fpage>586</fpage><lpage>592</lpage><pub-id pub-id-type="doi">10.1007/s11604-024-01712-2</pub-id><pub-id pub-id-type="medline">39625594</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jung</surname><given-names>J</given-names> </name><name name-style="western"><surname>Phillipi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Tran</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Accuracy of large language models in generating differential diagnosis from clinical presentation and imaging findings in pediatric cases</article-title><source>Pediatr Radiol</source><year>2025</year><month>08</month><volume>55</volume><issue>9</issue><fpage>1927</fpage><lpage>1933</lpage><pub-id pub-id-type="doi">10.1007/s00247-025-06317-z</pub-id><pub-id pub-id-type="medline">40650735</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>RP</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>WJ</given-names> </name><etal/></person-group><article-title>Evaluation of large language models for diagnostic impression generation from brain MRI report findings: a multicenter benchmark and reader study</article-title><source>NPJ Digit Med</source><year>2026</year><month>01</month><day>22</day><volume>9</volume><issue>1</issue><fpage>41571872</fpage><pub-id pub-id-type="doi">10.1038/s41746-026-02380-4</pub-id><pub-id pub-id-type="medline">41571872</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>D</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><etal/></person-group><article-title>DeepSeek-R1 incentivizes reasoning in LLMs through reinforcement learning</article-title><source>Nature</source><year>2025</year><month>09</month><volume>645</volume><issue>8081</issue><fpage>633</fpage><lpage>638</lpage><pub-id pub-id-type="doi">10.1038/s41586-025-09422-z</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><article-title>Open-source LLM DeepSeek on a par with proprietary models in clinical decision making</article-title><source>Nat Med</source><year>2025</year><month>08</month><volume>31</volume><issue>8</issue><fpage>2496</fpage><lpage>2497</lpage><pub-id pub-id-type="doi">10.1038/s41591-025-03850-0</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>JianBo</surname><given-names>C</given-names> </name><name name-style="western"><surname>HouShi</surname><given-names>X</given-names> </name><name name-style="western"><surname>JunJi</surname><given-names>W</given-names> </name></person-group><article-title>The advantage of DeepSeek in local deployment technology bridges the gap between research and practice</article-title><source>Int J Surg</source><year>2025</year><volume>111</volume><issue>10</issue><fpage>7401</fpage><lpage>7402</lpage><pub-id pub-id-type="doi">10.1097/JS9.0000000000002784</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Han</surname><given-names>D</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>He</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Differentiating mass-forming intrahepatic cholangiocarcinoma from atypical hepatocellular carcinoma using Gd-EOB-DTPA-enhanced magnetic resonance imaging combined with serum markers in at-risk patients with hepatitis B virus</article-title><source>Quant Imaging Med Surg</source><year>2023</year><month>10</month><day>1</day><volume>13</volume><issue>10</issue><fpage>7156</fpage><lpage>7169</lpage><pub-id pub-id-type="doi">10.21037/qims-23-396</pub-id><pub-id pub-id-type="medline">37869332</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>SH</given-names> </name><name name-style="western"><surname>Schramm</surname><given-names>S</given-names> </name><name name-style="western"><surname>Adams</surname><given-names>LC</given-names> </name><etal/></person-group><article-title>Benchmarking the diagnostic performance of open source LLMs in 1933 Eurorad case reports</article-title><source>NPJ Digit Med</source><year>2025</year><month>02</month><day>12</day><volume>8</volume><issue>1</issue><fpage>97</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01488-3</pub-id><pub-id pub-id-type="medline">39934372</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Banerjee</surname><given-names>I</given-names> </name><name name-style="western"><surname>Bhattacharjee</surname><given-names>K</given-names> </name><name name-style="western"><surname>Burns</surname><given-names>JL</given-names> </name><etal/></person-group><article-title>&#x201C;Shortcuts&#x201D; causing bias in radiology artificial intelligence: causes, evaluation, and mitigation</article-title><source>J Am Coll Radiol</source><year>2023</year><month>09</month><volume>20</volume><issue>9</issue><fpage>842</fpage><lpage>851</lpage><pub-id pub-id-type="doi">10.1016/j.jacr.2023.06.025</pub-id><pub-id pub-id-type="medline">37506964</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rueckel</surname><given-names>J</given-names> </name><name name-style="western"><surname>Huemmer</surname><given-names>C</given-names> </name><name name-style="western"><surname>Fieselmann</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Pneumothorax detection in chest radiographs: optimizing artificial intelligence system for accuracy and confounding bias reduction using in-image annotations in algorithm training</article-title><source>Eur Radiol</source><year>2021</year><month>10</month><volume>31</volume><issue>10</issue><fpage>7888</fpage><lpage>7900</lpage><pub-id pub-id-type="doi">10.1007/s00330-021-07833-w</pub-id><pub-id pub-id-type="medline">33774722</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>LoPiccolo</surname><given-names>J</given-names> </name><name name-style="western"><surname>Gusev</surname><given-names>A</given-names> </name><name name-style="western"><surname>Christiani</surname><given-names>DC</given-names> </name><name name-style="western"><surname>J&#x00E4;nne</surname><given-names>PA</given-names> </name></person-group><article-title>Lung cancer in patients who have never smoked - an emerging disease</article-title><source>Nat Rev Clin Oncol</source><year>2024</year><month>02</month><volume>21</volume><issue>2</issue><fpage>121</fpage><lpage>146</lpage><pub-id pub-id-type="doi">10.1038/s41571-023-00844-0</pub-id><pub-id pub-id-type="medline">38195910</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hong</surname><given-names>EK</given-names> </name><name name-style="western"><surname>Ham</surname><given-names>J</given-names> </name><name name-style="western"><surname>Roh</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Diagnostic accuracy and clinical value of a domain-specific multimodal generative AI model for chest radiograph report generation</article-title><source>Radiology</source><year>2025</year><month>03</month><volume>314</volume><issue>3</issue><fpage>e241476</fpage><pub-id pub-id-type="doi">10.1148/radiol.241476</pub-id><pub-id pub-id-type="medline">40131111</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zack</surname><given-names>T</given-names> </name><name name-style="western"><surname>Lehman</surname><given-names>E</given-names> </name><name name-style="western"><surname>Suzgun</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Assessing the potential of GPT-4 to perpetuate racial and gender biases in health care: a model evaluation study</article-title><source>Lancet Digit Health</source><year>2024</year><month>01</month><volume>6</volume><issue>1</issue><fpage>e12</fpage><lpage>e22</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(23)00225-X</pub-id><pub-id pub-id-type="medline">38123252</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cozzi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Pinker</surname><given-names>K</given-names> </name><name name-style="western"><surname>Hidber</surname><given-names>A</given-names> </name><etal/></person-group><article-title>BI-RADS category assignments by GPT-3.5, GPT-4, and Google Bard: a multilanguage study</article-title><source>Radiology</source><year>2024</year><month>04</month><volume>311</volume><issue>1</issue><fpage>e232133</fpage><pub-id pub-id-type="doi">10.1148/radiol.232133</pub-id><pub-id pub-id-type="medline">38687216</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Meddeb</surname><given-names>A</given-names> </name><name name-style="western"><surname>L&#x00FC;ken</surname><given-names>S</given-names> </name><name name-style="western"><surname>Busch</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Large language model ability to translate CT and MRI free-text radiology reports into multiple languages</article-title><source>Radiology</source><year>2024</year><month>12</month><volume>313</volume><issue>3</issue><fpage>e241736</fpage><pub-id pub-id-type="doi">10.1148/radiol.241736</pub-id><pub-id pub-id-type="medline">39688492</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Larson</surname><given-names>DB</given-names> </name><name name-style="western"><surname>Koirala</surname><given-names>A</given-names> </name><name name-style="western"><surname>Cheuy</surname><given-names>LY</given-names> </name><etal/></person-group><article-title>Assessing completeness of clinical histories accompanying imaging orders using adapted open-source and closed-source large language models</article-title><source>Radiology</source><year>2025</year><month>02</month><volume>314</volume><issue>2</issue><fpage>e241051</fpage><pub-id pub-id-type="doi">10.1148/radiol.241051</pub-id><pub-id pub-id-type="medline">39998369</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bhayana</surname><given-names>R</given-names> </name><name name-style="western"><surname>Alwahbi</surname><given-names>O</given-names> </name><name name-style="western"><surname>Ladak</surname><given-names>AM</given-names> </name><etal/></person-group><article-title>Leveraging large language models to generate clinical histories for oncologic imaging requisitions</article-title><source>Radiology</source><year>2025</year><month>02</month><volume>314</volume><issue>2</issue><fpage>e242134</fpage><pub-id pub-id-type="doi">10.1148/radiol.242134</pub-id><pub-id pub-id-type="medline">39903072</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Template examples for different diseases.</p><media xlink:href="jmir_v28i1e94904_app1.docx" xlink:title="DOCX File, 30 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Model parameters and radiologist characteristics.</p><media xlink:href="jmir_v28i1e94904_app2.docx" xlink:title="DOCX File, 17 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>All statistical values in the study.</p><media xlink:href="jmir_v28i1e94904_app3.docx" xlink:title="DOCX File, 40 KB"/></supplementary-material></app-group></back></article>