<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e93890</article-id><article-id pub-id-type="doi">10.2196/93890</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>The Performance of ChatGPT-4o and DeepSeek-R1 in Interpreting Thyroid Nodule Ultrasound Text Reports: Multicenter Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Xie</surname><given-names>Yujie</given-names></name><degrees>BMed</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Liu</surname><given-names>Jiarui</given-names></name><degrees>BMed</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhan</surname><given-names>Bing</given-names></name><degrees>BMed</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Kangfan</given-names></name><degrees>BMed</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Li</surname><given-names>Yuchen</given-names></name><degrees>MM</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Ning</surname><given-names>Chunping</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Ultrasound, Affiliated Hospital of Qingdao University</institution><addr-line>No. 16 Jiangsu Road</addr-line><addr-line>Qingdao</addr-line><addr-line>Shandong</addr-line><country>China</country></aff><aff id="aff2"><institution>Department of Ultrasound, JiaoZhou Central Hospital of Qingdao</institution><addr-line>Qingdao</addr-line><country>China</country></aff><aff id="aff3"><institution>Department of Ultrasound, Tai'an City Central Hospital</institution><addr-line>Tai'an</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Xu</surname><given-names>Weilin</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Qin</surname><given-names>Weisiyu</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Su</surname><given-names>Zichang</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Chunping Ning, MD, Department of Ultrasound, Affiliated Hospital of Qingdao University, No. 16 Jiangsu Road, Qingdao, Shandong, 266000, China, 86 18661806751; <email>152081340@qq.com</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>28</day><month>7</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e93890</elocation-id><history><date date-type="received"><day>21</day><month>02</month><year>2026</year></date><date date-type="rev-recd"><day>30</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>30</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Yujie Xie, Jiarui Liu, Bing Zhan, Kangfan Zhang, Yuchen Li, Chunping Ning. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 28.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e93890"/><abstract><sec><title>Background</title><p>Although thyroid nodules are detected in up to 60% of adults on ultrasound, the vast majority are benign, creating a substantial decision-making burden compounded by heterogeneous practice guidelines. Large language models (LLMs) show promise in processing unstructured medical text and are emerging as tools for report interpretation among both clinicians and patients. However, their reliability across distinct clinical tasks in thyroid ultrasound interpretation remains poorly characterized.</p></sec><sec><title>Objective</title><p>This study evaluates 2 LLMs, ChatGPT-4o and DeepSeek-R1, in interpreting thyroid nodule ultrasound text reports across three clinical tasks&#x2014;benign-malignant differentiation, Chinese Thyroid Imaging Reporting and Data System (C-TIRADS) classification, and management recommendation&#x2014;with concurrent assessment of output stability for each task.</p></sec><sec sec-type="methods"><title>Methods</title><p>We retrospectively analyzed 1063 ultrasound text reports from 3 medical centers, including 306 with histopathological confirmation. Each nodule report was submitted to both LLMs via their consumer web interfaces using task-specific prompts, with 5 repetitions per model; final outputs were determined by mode voting. Diagnostic performance was assessed by receiver operating characteristic analysis with DeLong testing; agreement was quantified using squared weighted &#x03BA; and Cohen &#x03BA;; and stability was measured using Krippendorff &#x03B1; and Fleiss &#x03BA;.</p></sec><sec sec-type="results"><title>Results</title><p>For benign-malignant differentiation, DeepSeek-R1 showed higher sensitivity (0.879 vs 0.692; <italic>P</italic>&#x003C;.001) and accuracy (0.729 vs 0.644; <italic>P</italic>=.008) than ChatGPT-4o. With access to images and clinical context unavailable to the LLMs, senior radiologists showed higher performance (area under the curve=0.865; accuracy=0.804). For C-TIRADS classification, DeepSeek-R1 showed substantial agreement with radiologists, exceeding ChatGPT-4o (&#x03BA;=0.770 vs 0.688; &#x0394;&#x03BA;=0.082, 95% CI 0.048-0.122). Both models yielded moderate, comparable agreement with clinicians on management recommendations (&#x03BA;=0.606 vs 0.608). Stability was near perfect for C-TIRADS classification (&#x03B1;=0.864 vs 0.866) and management recommendations (&#x03BA;=0.853 vs. 0.849) in both models; however, DeepSeek-R1 showed markedly greater stability than ChatGPT-4o in benign-malignant differentiation (&#x03BA;=0.869 vs 0.609; &#x0394;&#x03BA;=0.260, 95% CI 0.191-0.321).</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Both LLMs demonstrate clinical potential for thyroid nodule ultrasound report interpretation, with DeepSeek-R1 showing advantages in diagnostic accuracy, classification consistency, and output stability. However, both LLMs remained inferior to senior radiologists, suggesting their role as decision-support tools rather than stand-alone diagnostic systems. These findings provide preliminary evidence to inform the responsible integration of LLMs into thyroid imaging workflows while highlighting the need for further evaluation before patient-facing deployment.</p></sec></abstract><kwd-group><kwd>large language model</kwd><kwd>artificial intelligence</kwd><kwd>ChatGPT</kwd><kwd>DeepSeek</kwd><kwd>thyroid nodule</kwd><kwd>ultrasound</kwd><kwd>Chinese Thyroid Imaging Reporting and Data System</kwd><kwd>C-TIRADS</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Thyroid nodules are among the most common endocrine findings in adults, detected in up to 60% of healthy individuals on ultrasound screening [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. However, 90% to 95% are histologically benign, and most malignancies follow an indolent clinical course [<xref ref-type="bibr" rid="ref4">4</xref>]. This combination of a high detection rate and a low malignancy risk imposes a substantial decision-making burden on clinicians, further compounded by considerable heterogeneity in diagnostic and management practices across guidelines, regions, and institutions [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>]. Notable variability persists in key clinical decisions, including surveillance intervals, indications for fine-needle aspiration (FNA), and the timing of intervention [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>].</p><p>Recent advances in AI&#x2014;particularly large language models (LLMs)&#x2014;have introduced a transformative approach for processing and interpreting unstructured medical data [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. Emerging evidence indicates that LLMs such as GPT-3.5 (OpenAI), GPT-4.0, and Microsoft Bing can automate the structured processing of imaging reports [<xref ref-type="bibr" rid="ref11">11</xref>-<xref ref-type="bibr" rid="ref13">13</xref>], extract diagnostic information from free-text content [<xref ref-type="bibr" rid="ref14">14</xref>], and respond to patient-facing medical queries [<xref ref-type="bibr" rid="ref15">15</xref>]. In thyroid imaging specifically, recent work has explored multimodal LLMs for ultrasound-based nodule classification [<xref ref-type="bibr" rid="ref16">16</xref>]. However, the role of LLMs in interpreting text-based thyroid ultrasound reports remains poorly defined, with their performance across different clinical tasks and the reliability of their outputs yet to be systematically characterized.</p><p>Against this backdrop, LLMs could play a meaningful role in 3 distinct scenarios. The first is structured interpretation, in which LLMs assign standardized risk categories such as Chinese Thyroid Imaging Reporting and Data System (C-TIRADS) based on textual sonographic descriptions. The second is clinical decision support, in which LLMs generate management recommendations to assist clinicians. The third is text-based diagnostic assessment, in which LLMs independently infer benign-malignant likelihood from report content&#x2014;a scenario of particular relevance as patients increasingly consult publicly accessible LLMs to interpret their own reports. Across all 3 scenarios, output stability remains a foundational prerequisite, given the inherent stochasticity of generative models.</p><p>In light of these gaps, we conducted a multicenter study evaluating 2 leading LLMs&#x2014;ChatGPT-4o and DeepSeek-R1&#x2014;across 3 corresponding tasks: (1) benign-malignant differentiation, (2) C-TIRADS classification, and (3) management recommendation, with concurrent assessment of output stability across repeated queries. By benchmarking LLM performance across these clinically meaningful dimensions, our findings aim to inform the professional use of LLMs in thyroid nodule ultrasound interpretation, with direct relevance to emerging patient-facing applications and to support the evidence-based integration of generative AI into clinical practice.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Ethical Considerations</title><p>This multicenter study was conducted in accordance with the Declaration of Helsinki and received approval from the ethics committee (QYFY WZLL 30331). The requirement for informed consent was waived due to the retrospective study design. Data input into LLMs were deidentified. No identifiable participant images are included in this manuscript or its supplementary materials. The overall study design is illustrated in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Overall study workflow. ChatGPT-4o and DeepSeek-R1 were evaluated on thyroid nodule ultrasound text reports for benign-malignant differentiation, C-TIRADS classification, and management recommendation. For each task, each model was queried five times per report using a task-specific prompt. Repeated outputs were used for 2 downstream analyses: stability assessment (middle panel) and task-specific evaluation based on the mode-voted final result (lower panel), with stability presented before performance in the revised workflow. C-TIRADS: Chinese Thyroid Imaging Reporting and Data System; LLMs: large language models; NPV: negative predictive value; PPV: positive predictive value.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e93890_fig01.png"/></fig></sec><sec id="s2-2"><title>Data</title><p>Ultrasound text reports of 1063 thyroid nodules were collected from 3 medical centers (A, B, and C). Center-level characteristics, including ultrasound equipment, the qualifications of radiologists and clinicians, and report templates, are summarized in Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>, while cross-center variations in reporting language are documented in Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. The cases were collected between January 2023 and December 2024. Of these, 306 nodules had definitive cytopathological or histopathological diagnoses. Inclusion criteria were as follows: (1) thyroid nodules examined by senior radiologists; (2) a single nodule or, in cases with multiple nodules, the most representative nodule, defined as the one with the highest C-TIRADS category&#x2014;if multiple nodules shared the same category, the largest nodule (by maximum diameter) was selected; and (3) a complete histological report for pathologically diagnosed cases. Exclusion criteria were as follows: (1) patients with other thyroid diseases; (2) incomplete sonographic descriptions of nodule characteristics or the radiologist&#x2019;s diagnostic conclusion; (3) unclear management recommendations; and (4) pathological results lacking a definitive benign or malignant classification, including cytological results corresponding to Bethesda category I (nondiagnostic), III (atypia of undetermined significance), IV (follicular neoplasm), or V (suspicious for malignancy), and histopathological reports with indeterminate diagnoses (eg, tumors of uncertain malignant potential).</p><p>Histopathological confirmation was available only for patients who underwent FNA or thyroidectomy based on clinical indications (eg, C-TIRADS &#x2265;4a, progressive enlargement, or patient preference), which inherently enriched this subset with higher-risk nodules.</p></sec><sec id="s2-3"><title>LLMS</title><p>Two LLMs, ChatGPT-4o and DeepSeek-R1 (671B parameters; Mixture-of-Experts architecture), were employed in this study. DeepSeek-R1 was used with the &#x201C;Deep Think&#x201D; mode enabled. Neither model underwent additional training on dedicated medical imaging datasets.</p></sec><sec id="s2-4"><title>Standardized Prompt</title><p>Consistent with clinical practice, all prompts were developed and administered in Chinese, with original ultrasound reports input directly without translation. A zero-shot prompting approach was adopted for all tasks. Each prompt constrained the output to a predefined categorical format. English translations, verified by 2 bilingual medical researchers, are provided for reference only (Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p></sec><sec id="s2-5"><title>Query Execution</title><p>All queries were performed manually by 3 trained operators from May 3 to June 27, 2025, during which no core model updates were released for either LLM. Both models were accessed through their official consumer-facing web interfaces (chat.openai.com and chat.deepseek.com); no browser-based automation, API calls, or third-party plugins were used. To ensure independence across successive queries, all operators followed a standardized operating procedure incorporating three safeguards: (1) each query was initiated in a newly opened chat session, (2) the conversation was manually deleted from the chat history immediately after response recording, and (3) no user-specific personalization features (eg, custom instructions or persistent memory) were enabled on either platform. Workload was evenly distributed across the 3 operators to minimize operator-specific bias.</p></sec><sec id="s2-6"><title>Experimental Program</title><sec id="s2-6-1"><title>Benign-Malignant Differentiation</title><p>Sonographic descriptions of 306 pathologically confirmed thyroid nodules were independently input into 2 LLMs five times each using standardized prompt 1. Given the inherent stochasticity of LLM outputs, a five-repetition design with mode voting was adopted to generate a stable consensus prediction, consistent with prior LLM-based medical studies [<xref ref-type="bibr" rid="ref17">17</xref>-<xref ref-type="bibr" rid="ref19">19</xref>]. The consensus diagnosis was compared with histopathology as the gold standard. Model stability was evaluated by comparing the 5 individual outputs for each LLM.</p></sec><sec id="s2-6-2"><title>C-TIRADS Classification</title><p>Sonographic descriptions were input into 2 LLMs five times each using standardized prompt 2. The final C-TIRADS category was determined by mode voting (as described above). When multiple modes were present, a tie-breaking rule was applied in which the highest category was selected to prioritize patient safety by minimizing the risk of underclassification. Agreement was gaged by comparing the ultimate classification with the senior radiologist&#x2019;s assessment, while stability was evaluated by examining the 5 outputs generated by each LLM.</p></sec><sec id="s2-6-3"><title>Management Recommendation</title><p>To simulate the real-world clinical workflow, the comprehensive report, including both the sonographic descriptions and the radiologist&#x2019;s C-TIRADS conclusion, was input into 2 LLMs five times each using standardized prompt 3. The output space was restricted to a binary choice (follow-up vs FNA), consistent with the initial ultrasound-based decision node defined in major thyroid nodule guidelines [<xref ref-type="bibr" rid="ref20">20</xref>-<xref ref-type="bibr" rid="ref22">22</xref>]; outputs recommending surgical referral were classified as format errors (Table S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The final recommendation was determined using mode voting (as described above) and assessed for consistency with the clinician&#x2019;s assessment. The stability of each LLM&#x2019;s 5 recommendations was evaluated through comparative analysis.</p><p>For benign-malignant differentiation, histopathological diagnosis served as the gold standard. For C-TIRADS classification, which lacks an objective ground truth, senior radiologists&#x2019; original assessments at each center served as the reference, with interrater variability minimized through the unified 2020 C-TIRADS guidelines and structured reporting templates (Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). For management recommendations, the documented decisions of attending clinicians served as the reference. Accordingly, performance on the latter 2 tasks should be interpreted as agreement with expert raters rather than diagnostic accuracy. Notably, LLMs were evaluated using text reports alone, whereas radiologists and clinicians had full access to ultrasound images and patient clinical information.</p></sec></sec><sec id="s2-7"><title>Nonstandard Output</title><p>All LLM outputs were independently reviewed by 2 investigators to identify nonstandard outputs, defined as any response that could not be directly mapped to a predefined answer category. Nonstandard outputs were classified into 3 mutually exclusive categories: format errors, equivocal responses, and hallucinations. Disagreements were resolved by consensus with a third investigator. Operational definitions and representative examples are provided in Table S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-8"><title>Statistical Analyses</title><p>Baseline comparability across the three centers was assessed using one-way ANOVA for age, the chi-square test for sex, and the Kruskal-Wallis test for C-TIRADS distribution. For the pathology-confirmed subset, between-center comparisons were performed using the Wilcoxon rank-sum test for age and C-TIRADS distribution, and the chi-square test for sex and malignancy proportion.</p><p>Diagnostic performance was evaluated using receiver operating characteristic (ROC) curve analysis with the area under the curve (AUC). For the senior radiologists, C-TIRADS categories served as the ordinal predictor for ROC construction. The consumer-facing web interfaces of ChatGPT-4o and DeepSeek-R1 do not expose native token-level probabilities [<xref ref-type="bibr" rid="ref23">23</xref>], precluding direct probabilistic ROC analysis. To ensure a methodologically equivalent comparison, the count of malignant verdicts (range: 0&#x2010;5) across 5 independent evaluation sessions served as the ordinal predictor of malignancy for the LLMs, in line with the self-consistency paradigm [<xref ref-type="bibr" rid="ref24">24</xref>], in which aggregating multiple independent inferences yields more reliable predictions than any single run [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]. Paired AUC comparisons were performed using the DeLong test on the same set of 306 pathologically confirmed nodules.</p><p>Sensitivity, specificity, positive predictive value (PPV), negative predictive value (NPV), accuracy, and <italic>F</italic><sub>1</sub>-score were calculated from mode-voting binary predictions for the LLMs. For the senior radiologists, the optimal binarization threshold was determined by maximizing the Youden index from the ROC curve, and the same metrics were calculated accordingly. Between-model comparisons of sensitivity, specificity, and accuracy were conducted using the McNemar test with continuity correction. For PPV, NPV, and <italic>F</italic><sub>1</sub>-score, the differences between the 2 models and their 95% CI were estimated using the bias-corrected and accelerated bootstrap method as described below.</p><p>The agreement of C-TIRADS classification between LLMs and radiologists was assessed using the squared weighted &#x03BA; coefficient [<xref ref-type="bibr" rid="ref26">26</xref>]; Cohen &#x03BA; coefficient was employed to evaluate agreement in management recommendations between LLMs and clinicians. To evaluate the robustness of the C-TIRADS classification results to the predefined higher-category tie-breaking rule, a sensitivity analysis was performed by alternatively assigning tied cases to the lower category, and the squared weighted &#x03BA; with senior radiologists was recalculated under both strategies.</p><p>Krippendorff &#x03B1; coefficient was used to assess the internal stability of C-TIRADS classification across repeated runs of the 2 LLMs. The stability of management recommendations and benign-malignant classification was evaluated using Fleiss &#x03BA; coefficient.</p><p>The choice of these agreement and stability indices followed standard practice: squared weighted &#x03BA; [<xref ref-type="bibr" rid="ref26">26</xref>] penalizes larger category discrepancies more heavily than smaller ones, thereby accounting for ordinal proximity in C-TIRADS categories; Cohen &#x03BA; is appropriate for binary agreement; Krippendorff &#x03B1; handles multicategory stability with multiple measurements; and Fleiss &#x03BA; extends to binary stability across repeated runs.</p><p>All 95% CIs for point estimates were calculated using the bias-corrected and accelerated bootstrap method with 5000 resamples (case-level resampling with replacement). For between-LLM comparisons of agreement or stability metrics, the 95% CI for the difference (DeepSeek-R1 &#x2212; ChatGPT-4o) was computed using the same procedure. For the tie-breaking sensitivity analysis, the between-strategy difference in weighted &#x03BA; was defined as &#x0394;&#x03BA;=&#x03BA;<sub>high</sub> &#x2212; &#x03BA;<sub>low</sub>. Differences were considered statistically significant when the CI excluded zero.</p></sec><sec id="s2-9"><title>Statistical Software and Interpretation Criteria</title><p>All statistical analyses were conducted using R software (version 4.4.3; R Foundation for Statistical Computing), with 2-sided <italic>P</italic>&#x003C;.05 considered statistically significant. The following thresholds were used to interpret &#x03BA; and &#x03B1; values, according to the criteria proposed by Landis and Koch [<xref ref-type="bibr" rid="ref27">27</xref>]: &#x2264;0.20, poor; 0.21&#x2010;0.40, fair; 0.41&#x2010;0.60, moderate; 0.61&#x2010;0.80, substantial; and 0.81&#x2010;1.00, almost perfect agreement.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Patient and Center Characteristics</title><p>A total of 1063 thyroid nodules from 3 centers&#x2014;Center A (n=448), Center B (n=307), and Center C (n=308)&#x2014;were included in this study. The cohort comprised 740 (69.6%) female and 323 (30.4%) male patients, with a mean age of 46.59 (SD 14.46) years (range: 4&#x2010;90 y). Baseline comparisons across centers revealed significant differences in age and C-TIRADS distribution (<italic>P</italic>&#x003C;.001), whereas sex distribution was comparable (<italic>P</italic>=.30; <xref ref-type="table" rid="table1">Table 1</xref>).</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Demographic and clinical characteristics of the study population<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Characteristic</td><td align="left" valign="top">Overall (N=1063)</td><td align="left" valign="top">Center A (n=448)</td><td align="left" valign="top">Center B (n=307)</td><td align="left" valign="top">Center C (n=308)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Age (years), mean (SD; range)</td><td align="left" valign="top">46.59 (14.46; 4-90)</td><td align="left" valign="top">43.58 (14.10; 4-85)</td><td align="left" valign="top">51.32 (14.26; 16-90)</td><td align="left" valign="top">46.26 (13.99; 11-81)</td><td align="char" char="." valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Sex, n (%)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="char" char="." valign="top">.30</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Male</td><td align="left" valign="top">323 (30.4)</td><td align="left" valign="top">135 (30.1)</td><td align="left" valign="top">85 (27.7)</td><td align="left" valign="top">103 (33.4)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Female</td><td align="left" valign="top">740 (69.6)</td><td align="left" valign="top">313 (69.9)</td><td align="left" valign="top">222 (72.3)</td><td align="left" valign="top">205 (66.6)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">C-TIRADS<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> category, n (%)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="char" char="." valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>1</td><td align="left" valign="top">47 (4.4)</td><td align="left" valign="top">30 (6.7)</td><td align="left" valign="top">7 (2.3)</td><td align="left" valign="top">10 (3.2)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>2</td><td align="left" valign="top">115 (10.8)</td><td align="left" valign="top">49 (10.9)</td><td align="left" valign="top">29 (9.4)</td><td align="left" valign="top">37 (12.0)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>3</td><td align="left" valign="top">278 (26.2)</td><td align="left" valign="top">75 (16.7)</td><td align="left" valign="top">121 (39.4)</td><td align="left" valign="top">82 (26.6)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>4a</td><td align="left" valign="top">285 (26.8)</td><td align="left" valign="top">131 (29.2)</td><td align="left" valign="top">92 (30.0)</td><td align="left" valign="top">62 (20.1)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>4b</td><td align="left" valign="top">186 (17.5)</td><td align="left" valign="top">78 (17.4)</td><td align="left" valign="top">39 (12.7)</td><td align="left" valign="top">69 (22.4)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>4c</td><td align="left" valign="top">111 (10.4)</td><td align="left" valign="top">53 (11.8)</td><td align="left" valign="top">18 (5.9)</td><td align="left" valign="top">40 (13.0)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>5</td><td align="left" valign="top">41 (3.9)</td><td align="left" valign="top">32 (7.1)</td><td align="left" valign="top">1 (0.3)</td><td align="left" valign="top">8 (2.6)</td><td align="left" valign="top"/></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Data are presented as mean (SD; range) for continuous variables and n (%) for categorical variables.</p></fn><fn id="table1fn2"><p><sup>b</sup>C-TIRADS: Chinese Thyroid Imaging Reporting and Data System.</p></fn></table-wrap-foot></table-wrap><p>Among the overall cohort, 306 nodules (from Centers A and B; Center C did not contribute pathology-confirmed cases) had histopathological confirmation, including 124 (40.5%) benign and 182 (59.5%) malignant lesions. The pathology-confirmed subset showed a significant difference in age between the 2 centers (<italic>P</italic>=.002), while sex and C-TIRADS distributions were comparable (<italic>P</italic>=.70 and <italic>P</italic>=.17, respectively). The malignancy rate was significantly higher in Center B than in Center A (59/84, 70.2% vs 123/222, 55.4%; <italic>P</italic>=.03; <xref ref-type="table" rid="table2">Table 2</xref>).</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Demographic and clinical characteristics of the pathology-confirmed subset for benign-malignant differentiation analysis<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Characteristic</td><td align="left" valign="bottom">Total (N=306)</td><td align="left" valign="bottom">Center A (n=222)</td><td align="left" valign="bottom">Center B (n=84)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Age (y), mean (SD; range)</td><td align="left" valign="top">45.83 (13.69; 4&#x2010;85)</td><td align="left" valign="top">44.40 (14.00; 4&#x2010;85)</td><td align="left" valign="top">49.70 (12.00; 17&#x2010;81)</td><td align="char" char="." valign="top">.002</td></tr><tr><td align="left" valign="top">Sex, n (%)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="char" char="." valign="top">.70</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Male</td><td align="left" valign="top">87 (28.4)</td><td align="left" valign="top">65 (29.3)</td><td align="left" valign="top">22 (26.2)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Female</td><td align="left" valign="top">219 (71.6)</td><td align="left" valign="top">157 (70.7)</td><td align="left" valign="top">62 (73.8)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">C-TIRADS<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup> category, n (%)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="char" char="." valign="top">.17</td></tr><tr><td align="char" char="." valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>3</td><td align="left" valign="top">25 (8.2)</td><td align="left" valign="top">18 (8.1)</td><td align="left" valign="top">7 (8.3)</td><td align="left" valign="top"/></tr><tr><td align="char" char="." valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>4a</td><td align="left" valign="top">95 (31.0)</td><td align="left" valign="top">66 (29.7)</td><td align="left" valign="top">29 (34.5)</td><td align="left" valign="top"/></tr><tr><td align="char" char="." valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>4b</td><td align="left" valign="top">84 (27.5)</td><td align="left" valign="top">61 (27.5)</td><td align="left" valign="top">23 (27.4)</td><td align="left" valign="top"/></tr><tr><td align="char" char="." valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>4c</td><td align="left" valign="top">73 (23.9)</td><td align="left" valign="top">48 (21.6)</td><td align="left" valign="top">25 (29.8)</td><td align="left" valign="top"/></tr><tr><td align="char" char="." valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>5</td><td align="left" valign="top">29 (9.5)</td><td align="left" valign="top">29 (13.1)</td><td align="left" valign="top">0 (0.0)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">Histopathological diagnosis, n (%)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="char" char="." valign="top">.03</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Benign</td><td align="left" valign="top">124 (40.5)</td><td align="left" valign="top">99 (44.6)</td><td align="left" valign="top">25 (29.8)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Malignant</td><td align="left" valign="top">182 (59.5)</td><td align="left" valign="top">123 (55.4)</td><td align="left" valign="top">59 (70.2)</td><td align="left" valign="top"/></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Data are presented as mean (SD; range) for continuous variables and n (%) for categorical variables. Center C did not contribute pathology-confirmed cases. C-TIRADS categories 1 and 2 are absent because such nodules lacked histopathological reference standards.</p></fn><fn id="table2fn2"><p><sup>b</sup>C-TIRADS: Chinese Thyroid Imaging Reporting and Data System.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Benign-Malignant Differentiation</title><p>Among the 306 pathology-confirmed nodules (malignancy prevalence: n=182, 59.5%), the diagnostic performance of both LLMs was compared with that of senior radiologists. The AUC of DeepSeek-R1 was 0.718 (95% CI 0.665-0.770), numerically higher than that of ChatGPT-4o (AUC=0.688, 95% CI 0.629-0.746), although the difference was not statistically significant (<italic>P</italic>=.34). Both models showed significantly lower diagnostic performance than senior radiologists, who achieved an AUC of 0.865 (95% CI 0.828-0.902; both <italic>P</italic>&#x003C;.001 vs the LLMs; <xref ref-type="fig" rid="figure2">Figure 2</xref>).</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Receiver operating characteristic curves for ChatGPT-4o, DeepSeek-R1, and senior radiologists. AUC: area under the curve.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e93890_fig02.png"/></fig><p>Using the optimal cutoff of C-TIRADS &#x2265;4b (Youden index=0.588), senior radiologists achieved the highest performance across most indicators (sensitivity 0.846, specificity 0.742, accuracy 0.804, PPV 0.828, NPV 0.767, and <italic>F</italic><sub>1</sub>-score 0.837). Compared with ChatGPT-4o, DeepSeek-R1 showed significantly higher sensitivity (0.879 vs 0.692; <italic>P</italic>&#x003C;.001), accuracy (0.729 vs 0.644; <italic>P</italic>=.008), and NPV (0.741 vs 0.559; &#x0394;=0.182, 95% CI 0.088-0.271), whereas specificity (0.508 vs 0.573) and PPV (0.724 vs 0.704) were comparable between the 2 models (<xref ref-type="fig" rid="figure3">Figure 3</xref>; Table S5 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>(A) Confusion matrix for ChatGPT-4o. (B) Confusion matrix for DeepSeek-R1. (C) Confusion matrix for senior radiologists. (D) Comparison of performance: ChatGPT-4o vs DeepSeek-R1 vs senior radiologists. NPV: negative predictive value; PPV: positive predictive value.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e93890_fig03.png"/></fig></sec><sec id="s3-3"><title>C-TIRADS Classification</title><p>DeepSeek-R1 demonstrated substantial agreement with radiologists on C-TIRADS classification, showing significantly higher concordance than ChatGPT-4o (0.770, 95% CI 0.742-0.796 vs 0.688, 95% CI 0.644-0.724; &#x0394;&#x03BA;=0.082, 95% CI 0.048-0.122). Across the 3 centers, DeepSeek-R1 demonstrated moderate to almost perfect agreement (&#x03BA; range: 0.686&#x2010;0.822). In contrast, ChatGPT-4o&#x2019;s agreement with radiologists varied considerably, ranging from almost perfect agreement in Center A (0.788, 95% CI 0.733-0.828) to moderate agreement in Centers B and C (&#x03BA; range: 0.508&#x2010;0.594; <xref ref-type="table" rid="table3">Table 3</xref>).</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Agreement between large language models and radiologists on Chinese Thyroid Imaging Reporting and Data System classification<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Center</td><td align="left" valign="bottom">Nodules, n</td><td align="left" valign="bottom">ChatGPT-4o weighted &#x03BA; (95% CI)</td><td align="left" valign="bottom">DeepSeek-R1 weighted &#x03BA; (95% CI)</td><td align="left" valign="bottom">&#x0394; weighted &#x03BA; (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">A</td><td align="left" valign="top">448</td><td align="left" valign="top">0.788 (0.733 to 0.828)</td><td align="left" valign="top">0.822 (0.780 to 0.855)</td><td align="left" valign="top">0.034 (&#x2212;0.010 to 0.084)</td></tr><tr><td align="left" valign="top">B</td><td align="left" valign="top">307</td><td align="left" valign="top">0.508 (0.403 to 0.585)</td><td align="left" valign="top">0.686 (0.638 to 0.730)</td><td align="left" valign="top">0.178 (0.102 to 0.282)</td></tr><tr><td align="left" valign="top">C</td><td align="left" valign="top">308</td><td align="left" valign="top">0.594 (0.503 to 0.669)</td><td align="left" valign="top">0.744 (0.685 to 0.791)</td><td align="left" valign="top">0.150 (0.087 to 0.228)</td></tr><tr><td align="left" valign="top">Overall</td><td align="left" valign="top">1063</td><td align="left" valign="top">0.688 (0.644 to 0.724)</td><td align="left" valign="top">0.770 (0.742 to 0.796)</td><td align="left" valign="top">0.082 (0.048 to 0.122)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Data are weighted &#x03BA; values, with 95% CIs in parentheses. &#x0394; weighted &#x03BA; represents the mean difference in &#x03BA; values (DeepSeek-R1 &#x2212; ChatGPT-4o).</p></fn></table-wrap-foot></table-wrap><p>The classification performance was further examined in <xref ref-type="fig" rid="figure4">Figure 4</xref>. Both models exhibited strong agreement with radiologists for category 2 and 5 nodules (range: 73.2%&#x2010;97.4%). Challenges were observed in classifying category 4 nodules, with agreement rates ranging from 13.5% to 46.0%. For category 1 nodules, DeepSeek-R1 achieved 100% agreement, while ChatGPT-4o achieved 29.8% agreement.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Agreement between large language models and radiologists on Chinese Thyroid Imaging Reporting and Data System classification: confusion matrix analysis. LLMs: large language models.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e93890_fig04.png"/></fig><p>Tie-breaking occurred in 5.0% (53/1063) of ChatGPT-4o classifications and 10.0% (106/1063) of DeepSeek-R1 classifications, with most events involving adjacent categories, particularly within the category 4 spectrum (Table S6 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). All observed ties were 2-way; no three-or-more-way ties or fully dispersed configurations were encountered in either model. Replacing the higher-category strategy with a lower-category alternative yielded minimal changes in agreement: ChatGPT-4o (&#x03BA;=0.688 vs 0.690; &#x0394;&#x03BA;=&#x2212;0.002, 95% CI &#x2212;0.016 to 0.013) and DeepSeek-R1 (&#x03BA;=0.770 vs 0.781; &#x0394;&#x03BA;=&#x2212;0.011, 95% CI &#x2212;0.019 to &#x2212;0.004), with both models maintaining substantial agreement under either strategy (Table S7 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p></sec><sec id="s3-4"><title>Management Recommendation</title><p>Across the 1063 thyroid nodules, both models showed moderate and comparable agreement with clinicians (ChatGPT-4o: &#x03BA;=0.608, 95% CI 0.557 to 0.654; DeepSeek-R1: &#x03BA;=0.606, 95% CI 0.557 to 0.653; &#x0394;&#x03BA;=&#x2212;0.002, 95% CI &#x2212;0.044 to 0.041). Center-level analysis showed significant differences only in Center B, where DeepSeek-R1 had higher agreement than ChatGPT-4o (&#x03BA;=0.474, 95% CI 0.359 to 0.578 vs 0.342, 95% CI 0.224 to 0.462; &#x0394;&#x03BA;=0.132, 95% CI 0.029 to 0.243; <xref ref-type="table" rid="table4">Table 4</xref>). The overall percentage agreement was similarly high for both models (ChatGPT-4o: 861, 81.0%; DeepSeek-R1: 860, 80.9%; <xref ref-type="fig" rid="figure5">Figure 5</xref>).</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Agreement between large language models and clinicians on management recommendations<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup>.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Center</td><td align="left" valign="bottom">Nodules, n</td><td align="left" valign="bottom">ChatGPT-4o &#x03BA; (95 % CI)</td><td align="left" valign="bottom">DeepSeek-R1 &#x03BA; (95% CI)</td><td align="left" valign="bottom">&#x0394; &#x03BA; (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">A</td><td align="left" valign="top">448</td><td align="left" valign="top">0.642 (0.566 to 0.713)</td><td align="left" valign="top">0.590 (0.515 to 0.665)</td><td align="left" valign="top">&#x2212;0.052 (&#x2212;0.109 to 0.002)</td></tr><tr><td align="left" valign="top">B</td><td align="left" valign="top">307</td><td align="left" valign="top">0.342 (0.224 to 0.462)</td><td align="left" valign="top">0.474 (0.359 to 0.578)</td><td align="left" valign="top">0.132 (0.029 to 0.243)</td></tr><tr><td align="left" valign="top">C</td><td align="left" valign="top">308</td><td align="left" valign="top">0.670 (0.583 to 0.749)</td><td align="left" valign="top">0.681 (0.594 to 0.758)</td><td align="left" valign="top">0.011 (&#x2212;0.064 to 0.085)</td></tr><tr><td align="left" valign="top">Overall</td><td align="left" valign="top">1063</td><td align="left" valign="top">0.608 (0.557 to 0.654)</td><td align="left" valign="top">0.606 (0.557 to 0.653)</td><td align="left" valign="top">&#x2212;0.002 (&#x2212;0.044 to 0.041)</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Data are Cohen &#x03BA; values, with 95% CIs in parentheses. &#x0394; &#x03BA; represents the mean difference in &#x03BA; values (DeepSeek-R1 &#x2212; ChatGPT-4o).</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Agreement between large language models and clinicians on management recommendations: confusion matrix analysis. FNA: fine-needle aspiration.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e93890_fig05.png"/></fig></sec><sec id="s3-5"><title>Output Stability</title><p>In benign-malignant differentiation, DeepSeek-R1 exhibited almost perfect stability (&#x03BA;=0.869, 95% CI 0.822-0.906), significantly higher than that of ChatGPT-4o (&#x03BA;=0.609, 95% CI 0.557-0.661; &#x0394;&#x03BA;=0.260, 95% CI 0.191-0.321).</p><p>The overall stability of C-TIRADS classification was almost perfect and comparable between the 2 models (ChatGPT-4o: &#x03B1;=0.866, 95% CI 0.851 to 0.879; DeepSeek-R1: &#x03B1;=0.864, 95% CI 0.849 to 0.879; &#x0394;&#x03B1;=&#x2212;0.002, 95% CI &#x2212;0.023 to 0.017; <xref ref-type="table" rid="table5">Table 5</xref>). However, category-specific analysis revealed distinct patterns. For category 1 nodules, DeepSeek-R1 achieved perfect stability (&#x03B1;=1.000), significantly higher than ChatGPT-4o (&#x03B1;=0.562, 95% CI 0.395 to 0.717; &#x0394;&#x03B1;=0.438, 95% CI 0.297 to 0.617). Conversely, ChatGPT-4o outperformed DeepSeek-R1 in categories 3, 4a, 4b, and 4c (category 3: &#x03B1;=0.789, 95% CI 0.749 to 0.822 vs 0.661, 95% CI 0.608 to 0.711; &#x0394;&#x03B1;=&#x2212;0.128, 95% CI &#x2212;0.193 to &#x2212;0.065]; category 4a: &#x03B1;=0.692, 95% CI 0.643 to 0.740 vs 0.574, 95% CI 0.517 to 0.630; &#x0394;&#x03B1;=&#x2212;0.118, 95% CI &#x2212;0.193 to &#x2212;0.044; category 4b: &#x03B1;=0.681, 95% CI 0.610 to 0.748 vs 0.449, 95% CI 0.377 to 0.528; &#x0394;&#x03B1;=&#x2212;0.232, 95% CI &#x2212;0.334 to &#x2212;0.130; category 4c: &#x03B1;=0.766, 95% CI 0.692 to 0.827 vs 0.490, 95% CI 0.398 to 0.597; &#x0394;&#x03B1;=&#x2212;0.276, 95% CI &#x2212;0.397 to &#x2212;0.155).</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Stability of ChatGPT-4o and DeepSeek-R1 in Chinese Thyroid Imaging Reporting and Data System classification<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup>.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Category</td><td align="left" valign="bottom">Nodules, n</td><td align="left" valign="bottom">ChatGPT-4o &#x03B1; (95 % CI)</td><td align="left" valign="bottom">DeepSeek-R1 &#x03B1; (95% CI)</td><td align="left" valign="bottom">&#x0394;&#x03B1; (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">Overall</td><td align="left" valign="top">1063</td><td align="left" valign="top">0.866 (0.851 to 0.879)</td><td align="left" valign="top">0.864 (0.849 to 0.879)</td><td align="left" valign="top">&#x2212;0.002 (&#x2212;0.023 to 0.017)</td></tr><tr><td align="left" valign="top">1</td><td align="left" valign="top">47</td><td align="left" valign="top">0.562 (0.395 to 0.717)</td><td align="left" valign="top">1.000 (&#x2014;<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup>)</td><td align="left" valign="top">0.438 (0.297 to 0.617)</td></tr><tr><td align="left" valign="top">2</td><td align="left" valign="top">115</td><td align="left" valign="top">0.627 (0.514 to 0.736)</td><td align="left" valign="top">0.416 (0.119 to 0.721)</td><td align="left" valign="top">&#x2212;0.211 (&#x2212;0.661 to 0.069)</td></tr><tr><td align="left" valign="top">3</td><td align="left" valign="top">278</td><td align="left" valign="top">0.789 (0.749 to 0.822)</td><td align="left" valign="top">0.661 (0.608 to 0.711)</td><td align="left" valign="top">&#x2212;0.128 (&#x2212;0.193 to &#x2212;0.065)</td></tr><tr><td align="left" valign="top">4a</td><td align="left" valign="top">285</td><td align="left" valign="top">0.692 (0.643 to 0.740)</td><td align="left" valign="top">0.574 (0.517 to 0.630)</td><td align="left" valign="top">&#x2212;0.118 (&#x2212;0.193 to &#x2212;0.044)</td></tr><tr><td align="left" valign="top">4b</td><td align="left" valign="top">186</td><td align="left" valign="top">0.681 (0.610 to 0.748)</td><td align="left" valign="top">0.449 (0.377 to 0.528)</td><td align="left" valign="top">&#x2212;0.232 (&#x2212;0.334 to &#x2212;0.130)</td></tr><tr><td align="left" valign="top">4c</td><td align="left" valign="top">111</td><td align="left" valign="top">0.766 (0.692 to 0.827)</td><td align="left" valign="top">0.490 (0.398 to 0.597)</td><td align="left" valign="top">&#x2212;0.276 (&#x2212;0.397 to &#x2212;0.155)</td></tr><tr><td align="left" valign="top">5</td><td align="left" valign="top">41</td><td align="left" valign="top">0.460 (0.282 to 0.668)</td><td align="left" valign="top">0.446 (0.275 to 0.671)</td><td align="left" valign="top">&#x2212;0.014 (&#x2212;0.287 to 0.277)</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>Data are Krippendorff &#x03B1; values, with 95% CIs in parentheses. &#x0394;&#x03B1; represents the mean difference in &#x03B1; values (DeepSeek-R1 &#x2212; ChatGPT-4o).</p></fn><fn id="table5fn2"><p><sup>b</sup>Not applicable.</p></fn></table-wrap-foot></table-wrap><p>Both models demonstrated almost perfect and comparable stability in generating management recommendations (ChatGPT-4o: &#x03BA;=0.849, 95% CI 0.826 to 0.870; DeepSeek-R1: &#x03BA;=0.853, 95% CI 0.830 to 0.874; &#x0394;&#x03BA;=0.004, 95% CI &#x2212;0.026 to 0.033).</p></sec><sec id="s3-6"><title>Nonstandard Output</title><p>Of the 12,160 responses generated by each model, nonstandard outputs accounted for 169 (1.39%) in ChatGPT-4o and 80 (0.66%) in DeepSeek-R1. Hallucinations, the category most directly relevant to patient safety, accounted for 81 (0.67%) and 35 (0.29%) of outputs in ChatGPT-4o and DeepSeek-R1, respectively. Across both models, nonstandard outputs were most frequent in the benign-malignant differentiation task and least frequent in the management recommendation task. Detailed distributions by category and task are provided in Table S8 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>In this multicenter study, we systematically evaluated 2 prominent LLMs, DeepSeek-R1 and ChatGPT-4o, in interpreting thyroid nodule ultrasound text reports across 3 clinically relevant tasks&#x2014;benign-malignant differentiation, C-TIRADS classification, and management recommendation&#x2014;with concurrent assessment of output stability. Our key findings include (1) DeepSeek-R1 demonstrated higher sensitivity (0.879 vs 0.692; <italic>P</italic>&#x003C;.001) and accuracy (0.729 vs 0.644; <italic>P</italic>=.008) than ChatGPT-4o in benign-malignant differentiation, though both remained inferior to senior radiologists (AUC=0.865; accuracy=0.804), (2) DeepSeek-R1 showed substantially higher agreement with radiologists in C-TIRADS classification (&#x03BA;=0.770 vs 0.688; &#x0394;&#x03BA;=0.082, 95% CI 0.048 to 0.122), (3) both models achieved moderate and comparable agreement with clinicians on management recommendations (&#x03BA;=0.606 vs 0.608), and (4) while both models demonstrated near-perfect stability for C-TIRADS classification (&#x03B1;=0.864 vs 0.866) and management recommendations (&#x03BA;=0.853 vs 0.849), DeepSeek-R1 exhibited markedly greater stability than ChatGPT-4o in benign-malignant differentiation (&#x03BA;=0.869 vs 0.609; &#x0394;&#x03BA;=0.260, 95% CI 0.191 to 0.321). These findings provide empirical evidence for the responsible integration of LLMs into thyroid imaging workflows while highlighting task-dependent performance limitations.</p></sec><sec id="s4-2"><title>Task-Specific Performance and Comparison</title><sec id="s4-2-1"><title>Interpretive Considerations</title><p>A conceptual distinction warrants emphasis before interpreting task-specific performance. For benign-malignant differentiation, histopathology served as the reference standard, enabling assessment of diagnostic accuracy. For C-TIRADS classification and management recommendations, the reference was human expert judgment; performance in these domains therefore reflects clinical concordance rather than objective correctness. As noted in the Introduction, considerable interclinician variability exists in thyroid nodule diagnosis and management. High agreement with expert raters indicates the model&#x2019;s capacity to replicate specific practice patterns but does not inherently validate clinical correctness, as human assessments remain fallible. Ultimate validation must rest on pathological outcomes and long-term patient prognosis rather than expert consensus alone.</p></sec><sec id="s4-2-2"><title>Benign-Malignant Differentiation</title><p>In the pathology-confirmed subset (n=306), DeepSeek-R1 demonstrated higher sensitivity (0.879 vs 0.692; <italic>P</italic>&#x003C;.001) and accuracy (0.729 vs 0.644; <italic>P</italic>=.008) than ChatGPT-4o, though the AUC difference was not statistically significant (0.718 vs 0.688; <italic>P</italic>=.34). With both models approaching the conventionally acceptable diagnostic range (AUC=0.7&#x2010;0.8) [<xref ref-type="bibr" rid="ref28">28</xref>], these statistically detectable differences should not be conflated with clinically meaningful superiority [<xref ref-type="bibr" rid="ref29">29</xref>]. Senior radiologists achieved higher performance (AUC=0.865); however, this comparison reflects informational asymmetry rather than inherent model inferiority, as radiologists accessed ultrasound images and clinical context, whereas LLMs operated on text reports alone. This gap suggests a potential auxiliary role for LLMs in image-limited scenarios such as remote consultation, patient self-interpretation of reports, and preliminary triage in primary care [<xref ref-type="bibr" rid="ref30">30</xref>]. Future work should prioritize multimodal medical AI [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref31">31</xref>] to enable more equitable comparisons.</p><p>The malignancy rate in our pathology-confirmed subset (182/306, 59.5%) substantially exceeded the general population prevalence (5%&#x2010;10%) [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref32">32</xref>], reflecting spectrum bias inherent to retrospective histopathology-based studies [<xref ref-type="bibr" rid="ref33">33</xref>]. This enrichment artificially inflates PPV and overall accuracy while potentially underestimating NPV; sensitivity, specificity, and AUC, though mathematically prevalence independent, may be indirectly affected by the narrower disease spectrum [<xref ref-type="bibr" rid="ref34">34</xref>]. Bayesian prevalence adjustment assuming real-world malignancy rates of 5% and 10% substantially reduced PPVs (ChatGPT-4o: 70.4%&#x2192;7.9%-15.3%; DeepSeek-R1: 72.4%&#x2192;8.6%-16.6%), while adjusted NPVs all exceeded 94% (Table S9 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). These findings suggest that both models retain strong rule-out capability in low-prevalence settings, though positive predictions require histopathological confirmation.</p></sec><sec id="s4-2-3"><title>C-TIRADS Classification</title><p>To our knowledge, this study represents the first application of LLMs to C-TIRADS classification based solely on sonographic descriptions. DeepSeek-R1 exhibited substantially higher concordance with senior radiologists than ChatGPT-4o (&#x03BA;=0.770 vs 0.688). This difference may relate to architectural distinctions: DeepSeek-R1 uses a Mixture-of-Experts architecture with 671 billion total parameters and 37 billion activated per token [<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref36">36</xref>], potentially enabling task-specific specialization [<xref ref-type="bibr" rid="ref37">37</xref>], combined with multistage reinforcement learning that may enhance structured reasoning [<xref ref-type="bibr" rid="ref38">38</xref>]. In contrast, GPT-4o, developed as a general-purpose multimodal model [<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref40">40</xref>], has not disclosed architectural specifics, limiting direct comparison. Moreover, the activated reasoning mode in DeepSeek-R1 may also contribute to the observed gap beyond architecture alone. These considerations remain speculative, as our black-box evaluation precluded analysis of internal mechanisms. Controlled architectural comparisons under matched inference configurations are warranted in future studies.</p></sec><sec id="s4-2-4"><title>Management Recommendation</title><p>Both LLMs demonstrated high raw agreement with clinicians (approximately 81%), while &#x03BA; values were moderate (&#x03BA;=0.606 vs 0.608). This approximately 20-percentage-point discrepancy primarily reflects &#x03BA;&#x2019;s mathematical properties, which account for chance agreement and are influenced by marginal category distributions [<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref42">42</xref>]. In datasets with imbalanced distributions&#x2014;such as ours, where C-TIRADS category proportions ranged from 3.9% to 26.8%&#x2014;the elevated baseline probability of chance agreement suppresses &#x03BA; relative to observed concordance [<xref ref-type="bibr" rid="ref42">42</xref>]. This well-documented statistical phenomenon does not imply inadequate performance [<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref42">42</xref>]; rather, our moderate-to-substantial &#x03BA; values (0.61&#x2010;0.77) indicate promising concordance requiring prospective validation before clinical deployment.</p><p>Performance varied across management categories. For follow-up recommendations, LLM concordance was relatively high (approximately 85.5%), whereas for invasive procedures such as FNA, agreement declined to 74.7%&#x2010;74.9%. This pattern aligns with prior findings that LLM accuracy in oncologic management decision support ranges from 50% to 70% [<xref ref-type="bibr" rid="ref43">43</xref>], underscoring proficiency in straightforward cases but inadequacy in complex decision-making [<xref ref-type="bibr" rid="ref44">44</xref>-<xref ref-type="bibr" rid="ref47">47</xref>]. LLMs face challenges in high-risk decisions due to difficulty incorporating individual patient factors&#x2014;age, comorbidities, clinical symptoms, and socioeconomic status&#x2014;essential to clinical judgment. Furthermore, they may not account for clinician caution under uncertainty or resource variability across settings [<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref49">49</xref>]. While proficient in standardized tasks, LLMs should serve as decision-support adjuncts rather than substitutes for physician judgment in complex treatment scenarios.</p></sec><sec id="s4-2-5"><title>Output Stability</title><p>Both models demonstrated near-perfect stability for C-TIRADS classification and management recommendations, supporting their reliability in structured, rule-based tasks. However, a marked disparity emerged in benign-malignant differentiation: DeepSeek-R1 maintained high stability (&#x03BA;=0.869), whereas ChatGPT-4o exhibited substantially lower consistency (&#x03BA;=0.609). This instability in ChatGPT-4o is particularly concerning in patient-facing contexts, where repeated queries on identical reports could yield divergent diagnostic impressions, generating unwarranted anxiety or misplaced reassurance. These findings contrast with prior reports of high ChatGPT reproducibility in tasks such as Response Evaluation Criteria In Solid Tumours (RECIST) assessment [<xref ref-type="bibr" rid="ref9">9</xref>], underscoring the absence of universally reliable performance across clinical contexts. The observed variability across task complexities suggests that stability should be evaluated as a task-specific property rather than a global model attribute, reinforcing the need for systematic benchmarking before clinical deployment.</p></sec></sec><sec id="s4-3"><title>Clinical Implications</title><p>The differential performance across tasks observed in our study suggests task-specific deployment strategies for LLMs in thyroid imaging workflows. For standardized tasks such as C-TIRADS classification, LLMs&#x2014;particularly DeepSeek-R1&#x2014;could serve as quality assurance tools to enhance reporting consistency and assist less experienced readers in adhering to structured reporting systems. For low-risk nodules requiring surveillance, LLMs might be explored as triage-support tools in primary care settings, though their actual impact on referral patterns requires prospective evaluation. For patient-facing applications, where patients increasingly consult publicly accessible LLMs to interpret their own reports, models might provide preliminary interpretations with explicit uncertainty quantification. However, the safety and benefit of such patient-facing use require dedicated evaluation before broad deployment.</p><p>Safety considerations are integral to clinical deployment. Encouragingly, both models exhibited low hallucination rates (0.29%&#x2010;0.67%)&#x2014;markedly lower than the 18% to 51% reported in less constrained clinical settings [<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref51">51</xref>]&#x2014;likely reflecting structured prompts and standardized C-TIRADS-aligned descriptors that narrowed the response space [<xref ref-type="bibr" rid="ref52">52</xref>]. Hallucinations were nonetheless most frequent in benign-malignant differentiation, consistent with the greater inferential burden of probabilistic binary judgments. Given that LLMs may deliver erroneous content with unwarranted confidence [<xref ref-type="bibr" rid="ref53">53</xref>], these findings support their deployment in similarly structured interpretation tasks as physician-supervised decision-support tools rather than autonomous diagnostic systems. Furthermore, regulatory frameworks for AI-assisted medical decision-making remain nascent, and institutional policies must address liability, informed consent, and clinical oversight before widespread deployment.</p></sec><sec id="s4-4"><title>Limitations</title><p>Several limitations merit consideration. First, LLMs were evaluated exclusively on Chinese-language text reports, whereas senior radiologists accessed ultrasound images and clinical context; multimodal validation is warranted for fairer comparisons. Second, histopathological confirmation was obtained only for clinically indicated nodules (306/1063, 28.8%), yielding a malignancy-enriched cohort (182/306, 59.5%) that may inflate diagnostic performance through spectrum bias; although Bayesian prevalence adjustment partially mitigates this issue, prospective validation in unselected screening populations is required. For nodules without histopathology, reference standards relied on radiologist assessment and clinical follow-up, which are imperfect. Third, prompt design may have influenced model performance. The management recommendation task incorporated the radiologist-derived C-TIRADS classification as model input; accordingly, performance on this task reflects the models&#x2019; adherence to management guidelines conditional on a human-derived classification, rather than their independent capacity to derive recommendations from the sonographic description alone. The binary output design may have penalized clinically defensible surgical referral recommendations, although such outputs were rare in our cohort (ChatGPT-4o: 21/12,160, 0.17%; DeepSeek-R1: 13/12,160, 0.11%); multitiered output schemes warrant exploration in future work. In addition, each task used a single standardized prompt without embedded guideline-based criteria; future studies should therefore explore diverse prompting strategies to isolate the autonomous reasoning capacity of LLMs. Fourth, both models were accessed through their consumer web interfaces, with DeepSeek-R1 querying the &#x201C;Deep Think&#x201D; reasoning mode and ChatGPT-4o using its standard chat configuration. Although this reflects common real-world end-user usage, this configuration asymmetry may have partially contributed to the observed performance differences. Reproducibility under standardized API configurations remains to be established. Fifth, the low hallucination rates observed under our constrained categorical output design should not be extrapolated to open-ended clinical queries, patient-facing free-text dialogue, or unconstrained generative tasks, where substantially higher rates have been reported [<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref51">51</xref>]; dedicated evaluation is warranted before broader deployment. Sixth, the vote-count surrogate used for LLM ROC construction should not be interpreted as a calibrated probability of malignancy. Vote dispersion across independent runs reflects interinference agreement rather than a true probabilistic confidence estimate, and its formal correspondence with native token-level probabilities warrants validation through API-based access in future work. Finally, all reports and prompts were in Chinese; DeepSeek-R1, trained on a larger Chinese-language corpus, may have benefited from language-specific optimization [<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref55">55</xref>], although this advantage was not uniformly observed across all tasks. Combined with the single-province recruitment and retrospective output characterization noted above, these factors limit geographic, linguistic, and real-world generalizability; prospective multiregional and multilingual validation is warranted.</p></sec><sec id="s4-5"><title>Future Directions</title><p>Several avenues warrant future investigation. First, multimodal integration combining text reports with ultrasound images could narrow the performance gap with radiologists; recent advances in vision-language models provide the technical foundation for such systems. Second, prospective validation in real-world clinical workflows is essential to assess the impact on diagnostic accuracy, workflow efficiency, and patient outcomes beyond retrospective concordance metrics. Third, ensemble methods combining multiple LLMs or hybrid architectures integrating LLMs with traditional machine learning models may enhance robustness and mitigate individual model weaknesses by leveraging the complementary strengths observed between DeepSeek-R1 and ChatGPT-4o. Fourth, prompt engineering strategies&#x2014;including few-shot learning, chain-of-thought prompting, and guideline-embedded prompts&#x2014;should be systematically evaluated to optimize performance across clinical tasks. Fifth, longitudinal studies tracking model performance across software updates are needed, as LLM capabilities evolve rapidly, and current findings may not generalize to future versions. Finally, health economic evaluations and studies examining the integration of LLMs into clinical workflows&#x2014;including their impact on clinician workload, patient satisfaction, and health care costs&#x2014;are needed to inform evidence-based policy decisions on AI adoption in thyroid imaging.</p></sec><sec id="s4-6"><title>Conclusions</title><p>Advanced LLMs, particularly DeepSeek-R1, demonstrate meaningful capability in thyroid nodule ultrasound report interpretation, with consistent C-TIRADS classification, acceptable benign-malignant differentiation, and high output stability in structured tasks. However, limitations in high-risk decision-making and diagnostic inference stability&#x2014;particularly for ChatGPT-4o&#x2014;indicate that LLMs are better suited as decision-support adjuncts to enhance clinician efficiency rather than autonomous diagnostic systems. These findings support the responsible integration of LLMs into thyroid imaging workflows under appropriate physician oversight. Fully realizing the clinical potential of LLMs will require multimodal integration, prospective validation, and evolving regulatory frameworks.</p></sec></sec></body><back><ack><p>We would like to express our gratitude to all participating centers for providing thyroid nodule ultrasound text data and technical support essential to the completion of this multicenter study. All authors bear full responsibility for the study design, data collection, statistical analysis, interpretation of the results, and manuscript drafting. ChatGPT-4o and DeepSeek-R1 were used exclusively as research experimental objects for performance evaluation in the context of this study. No generative AI tools were used to generate scientific content, conduct data analysis, produce study results, or assist in manuscript composition or interpretation.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the Clinical Medicine and X Research Program of Affiliated Hospital of Qingdao University (QDFY+X2024133).</p></sec><sec><title>Data Availability</title><p>The datasets used and/or analyzed during this study are available from the corresponding author upon reasonable request.</p></sec></notes><fn-group><fn fn-type="con"><p>YX conceived and designed the study, conducted data acquisition, performed the statistical analysis, and wrote the main manuscript text. BZ and KZ contributed to data interpretation, conducted data acquisition, and provided substantial revisions to the manuscript. YL and JL assisted with the literature review and contributed to the interpretation of the results. CN supervised the study, provided critical revisions, and ensured the integrity and accuracy of the research process. All authors reviewed and approved the final manuscript.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AUC</term><def><p>area under the curve</p></def></def-item><def-item><term id="abb2">C-TIRADS</term><def><p>Chinese Thyroid Imaging Reporting and Data System</p></def></def-item><def-item><term id="abb3">FNA</term><def><p>fine-needle aspiration</p></def></def-item><def-item><term id="abb4">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb5">NPV</term><def><p>negative predictive value</p></def></def-item><def-item><term id="abb6">PPV</term><def><p>positive predictive value</p></def></def-item><def-item><term id="abb7">RECIST</term><def><p>Response Evaluation Criteria in Solid Tumors</p></def></def-item><def-item><term id="abb8">ROC</term><def><p>receiver operating characteristic</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Grani</surname><given-names>G</given-names> </name><name name-style="western"><surname>Sponziello</surname><given-names>M</given-names> </name><name name-style="western"><surname>Filetti</surname><given-names>S</given-names> </name><name name-style="western"><surname>Durante</surname><given-names>C</given-names> </name></person-group><article-title>Thyroid nodules: diagnosis and management</article-title><source>Nat Rev Endocrinol</source><year>2024</year><month>12</month><volume>20</volume><issue>12</issue><fpage>715</fpage><lpage>728</lpage><pub-id pub-id-type="doi">10.1038/s41574-024-01025-4</pub-id><pub-id pub-id-type="medline">39152228</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Durante</surname><given-names>C</given-names> </name><name name-style="western"><surname>Grani</surname><given-names>G</given-names> </name><name name-style="western"><surname>Lamartina</surname><given-names>L</given-names> </name><name name-style="western"><surname>Filetti</surname><given-names>S</given-names> </name><name name-style="western"><surname>Mandel</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Cooper</surname><given-names>DS</given-names> </name></person-group><article-title>The diagnosis and management of thyroid nodules: a review</article-title><source>JAMA</source><year>2018</year><month>03</month><day>6</day><volume>319</volume><issue>9</issue><fpage>914</fpage><lpage>924</lpage><pub-id pub-id-type="doi">10.1001/jama.2018.0898</pub-id><pub-id pub-id-type="medline">29509871</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gharib</surname><given-names>H</given-names> </name><name name-style="western"><surname>Papini</surname><given-names>E</given-names> </name><name name-style="western"><surname>Garber</surname><given-names>JR</given-names> </name><etal/></person-group><article-title>American Association of Clinical Endocrinologists, American College of Endocrinology, and Associazione Medici Endocrinologi medical guidelines for clinical practice for the diagnosis and management of thyroid nodules--2016 update</article-title><source>Endocr Pract</source><year>2016</year><month>05</month><volume>22</volume><issue>5</issue><fpage>622</fpage><lpage>639</lpage><pub-id pub-id-type="doi">10.4158/EP161208.GL</pub-id><pub-id pub-id-type="medline">27167915</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alexander</surname><given-names>EK</given-names> </name><name name-style="western"><surname>Cibas</surname><given-names>ES</given-names> </name></person-group><article-title>Diagnosis of thyroid nodules</article-title><source>Lancet Diabetes Endocrinol</source><year>2022</year><month>07</month><volume>10</volume><issue>7</issue><fpage>533</fpage><lpage>539</lpage><pub-id pub-id-type="doi">10.1016/S2213-8587(22)00101-2</pub-id><pub-id pub-id-type="medline">35752200</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Isik</surname><given-names>A</given-names> </name><name name-style="western"><surname>Firat</surname><given-names>D</given-names> </name><name name-style="western"><surname>Yilmaz</surname><given-names>I</given-names> </name><etal/></person-group><article-title>A survey of current approaches to thyroid nodules and thyroid operations</article-title><source>Int J Surg</source><year>2018</year><month>06</month><volume>54</volume><issue>Pt A</issue><fpage>100</fpage><lpage>104</lpage><pub-id pub-id-type="doi">10.1016/j.ijsu.2018.04.037</pub-id><pub-id pub-id-type="medline">29709542</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ha</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Baek</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Na</surname><given-names>DG</given-names> </name><etal/></person-group><article-title>Diagnostic performance of practice guidelines for thyroid nodules: thyroid nodule size versus biopsy rates</article-title><source>Radiology</source><year>2019</year><month>04</month><volume>291</volume><issue>1</issue><fpage>92</fpage><lpage>99</lpage><pub-id pub-id-type="doi">10.1148/radiol.2019181723</pub-id><pub-id pub-id-type="medline">30777805</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alexander</surname><given-names>EK</given-names> </name><name name-style="western"><surname>Doherty</surname><given-names>GM</given-names> </name><name name-style="western"><surname>Barletta</surname><given-names>JA</given-names> </name></person-group><article-title>Management of thyroid nodules</article-title><source>Lancet Diabetes Endocrinol</source><year>2022</year><month>07</month><volume>10</volume><issue>7</issue><fpage>540</fpage><lpage>548</lpage><pub-id pub-id-type="doi">10.1016/S2213-8587(22)00139-5</pub-id><pub-id pub-id-type="medline">35752201</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yoon</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Han</surname><given-names>K</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>EK</given-names> </name><name name-style="western"><surname>Moon</surname><given-names>HJ</given-names> </name><name name-style="western"><surname>Kwak</surname><given-names>JY</given-names> </name></person-group><article-title>Diagnosis and management of small thyroid nodules: a comparative study with six guidelines for thyroid nodules</article-title><source>Radiology</source><year>2017</year><month>05</month><volume>283</volume><issue>2</issue><fpage>560</fpage><lpage>569</lpage><pub-id pub-id-type="doi">10.1148/radiol.2016160641</pub-id><pub-id pub-id-type="medline">27805843</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tordjman</surname><given-names>M</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Yuce</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Comparative benchmarking of the DeepSeek large language model on medical tasks and clinical reasoning</article-title><source>Nat Med</source><year>2025</year><month>08</month><volume>31</volume><issue>8</issue><fpage>2550</fpage><lpage>2555</lpage><pub-id pub-id-type="doi">10.1038/s41591-025-03726-3</pub-id><pub-id pub-id-type="medline">40267969</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mukherjee</surname><given-names>P</given-names> </name><name name-style="western"><surname>Hou</surname><given-names>B</given-names> </name><name name-style="western"><surname>Lanfredi</surname><given-names>RB</given-names> </name><name name-style="western"><surname>Summers</surname><given-names>RM</given-names> </name></person-group><article-title>Feasibility of using the privacy-preserving large language model Vicuna for labeling radiology reports</article-title><source>Radiology</source><year>2023</year><month>10</month><volume>309</volume><issue>1</issue><fpage>e231147</fpage><pub-id pub-id-type="doi">10.1148/radiol.231147</pub-id><pub-id pub-id-type="medline">37815442</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Adams</surname><given-names>LC</given-names> </name><name name-style="western"><surname>Truhn</surname><given-names>D</given-names> </name><name name-style="western"><surname>Busch</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Leveraging GPT-4 for post hoc transformation of free-text radiology reports into structured reporting: a multilingual feasibility study</article-title><source>Radiology</source><year>2023</year><month>05</month><volume>307</volume><issue>4</issue><fpage>e230725</fpage><pub-id pub-id-type="doi">10.1148/radiol.230725</pub-id><pub-id pub-id-type="medline">37014240</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jiang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Xia</surname><given-names>S</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Transforming free-text radiology reports into structured reports using ChatGPT: a study on thyroid ultrasonography</article-title><source>Eur J Radiol</source><year>2024</year><month>06</month><volume>175</volume><fpage>111458</fpage><pub-id pub-id-type="doi">10.1016/j.ejrad.2024.111458</pub-id><pub-id pub-id-type="medline">38613868</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Wei</surname><given-names>M</given-names> </name><name name-style="western"><surname>Qin</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Harnessing large language models for structured reporting in breast ultrasound: a comparative study of Open AI (GPT-4.0) and Microsoft Bing (GPT-4)</article-title><source>Ultrasound Med Biol</source><year>2024</year><month>11</month><volume>50</volume><issue>11</issue><fpage>1697</fpage><lpage>1703</lpage><pub-id pub-id-type="doi">10.1016/j.ultrasmedbio.2024.07.007</pub-id><pub-id pub-id-type="medline">39138026</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fink</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Bischoff</surname><given-names>A</given-names> </name><name name-style="western"><surname>Fink</surname><given-names>CA</given-names> </name><etal/></person-group><article-title>Potential of ChatGPT and GPT-4 for data mining of free-text CT reports on lung cancer</article-title><source>Radiology</source><year>2023</year><month>09</month><volume>308</volume><issue>3</issue><fpage>e231362</fpage><pub-id pub-id-type="doi">10.1148/radiol.231362</pub-id><pub-id pub-id-type="medline">37724963</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rahsepar</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Tavakoli</surname><given-names>N</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>GHJ</given-names> </name><name name-style="western"><surname>Hassani</surname><given-names>C</given-names> </name><name name-style="western"><surname>Abtin</surname><given-names>F</given-names> </name><name name-style="western"><surname>Bedayat</surname><given-names>A</given-names> </name></person-group><article-title>How AI responds to common lung cancer questions: ChatGPT versus Google Bard</article-title><source>Radiology</source><year>2023</year><month>06</month><volume>307</volume><issue>5</issue><fpage>e230922</fpage><pub-id pub-id-type="doi">10.1148/radiol.230922</pub-id><pub-id pub-id-type="medline">37310252</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Chambara</surname><given-names>N</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Assessing the feasibility of ChatGPT-4o and Claude 3-Opus in thyroid nodule classification based on ultrasound images</article-title><source>Endocrine</source><year>2025</year><month>03</month><volume>87</volume><issue>3</issue><fpage>1041</fpage><lpage>1049</lpage><pub-id pub-id-type="doi">10.1007/s12020-024-04066-x</pub-id><pub-id pub-id-type="medline">39394537</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Qiu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Luo</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Chung</surname><given-names>KY</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>W</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>H</given-names> </name></person-group><article-title>Enhancing the readability of online pediatric cataract education materials: a comparative study of large language models</article-title><source>Transl Vis Sci Technol</source><year>2025</year><month>08</month><day>1</day><volume>14</volume><issue>8</issue><fpage>19</fpage><pub-id pub-id-type="doi">10.1167/tvst.14.8.19</pub-id><pub-id pub-id-type="medline">40824260</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xia</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hua</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Mei</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Clinical application potential of large language model: a study based on thyroid nodules</article-title><source>Endocrine</source><year>2025</year><month>01</month><volume>87</volume><issue>1</issue><fpage>206</fpage><lpage>213</lpage><pub-id pub-id-type="doi">10.1007/s12020-024-03981-3</pub-id><pub-id pub-id-type="medline">39080210</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>SH</given-names> </name><name name-style="western"><surname>Tong</surname><given-names>WJ</given-names> </name><name name-style="western"><surname>Li</surname><given-names>MD</given-names> </name><etal/></person-group><article-title>Collaborative enhancement of consistency and accuracy in US diagnosis of thyroid nodules using large language models</article-title><source>Radiology</source><year>2024</year><month>03</month><volume>310</volume><issue>3</issue><fpage>e232255</fpage><pub-id pub-id-type="doi">10.1148/radiol.232255</pub-id><pub-id pub-id-type="medline">38470237</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhou</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yin</surname><given-names>L</given-names> </name><name name-style="western"><surname>Wei</surname><given-names>X</given-names> </name><etal/></person-group><article-title>2020 Chinese guidelines for ultrasound malignancy risk stratification of thyroid nodules: the C-TIRADS</article-title><source>Endocrine</source><year>2020</year><month>11</month><volume>70</volume><issue>2</issue><fpage>256</fpage><lpage>279</lpage><pub-id pub-id-type="doi">10.1007/s12020-020-02441-y</pub-id><pub-id pub-id-type="medline">32827126</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tessler</surname><given-names>FN</given-names> </name><name name-style="western"><surname>Middleton</surname><given-names>WD</given-names> </name><name name-style="western"><surname>Grant</surname><given-names>EG</given-names> </name><etal/></person-group><article-title>ACR thyroid imaging, reporting and data system (TI-RADS): white paper of the ACR TI-RADS committee</article-title><source>J Am Coll Radiol</source><year>2017</year><month>05</month><volume>14</volume><issue>5</issue><fpage>587</fpage><lpage>595</lpage><pub-id pub-id-type="doi">10.1016/j.jacr.2017.01.046</pub-id><pub-id pub-id-type="medline">28372962</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Haugen</surname><given-names>BR</given-names> </name><name name-style="western"><surname>Alexander</surname><given-names>EK</given-names> </name><name name-style="western"><surname>Bible</surname><given-names>KC</given-names> </name><etal/></person-group><article-title>2015 American Thyroid Association management guidelines for adult patients with thyroid nodules and differentiated thyroid cancer: the American Thyroid Association Guidelines Task Force on Thyroid Nodules and Differentiated Thyroid Cancer</article-title><source>Thyroid</source><year>2016</year><month>01</month><volume>26</volume><issue>1</issue><fpage>1</fpage><lpage>133</lpage><pub-id pub-id-type="doi">10.1089/thy.2015.0020</pub-id><pub-id pub-id-type="medline">26462967</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Geng</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cai</surname><given-names>F</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Koeppl</surname><given-names>H</given-names> </name><name name-style="western"><surname>Nakov</surname><given-names>P</given-names> </name><name name-style="western"><surname>Gurevych</surname><given-names>I</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Duh</surname><given-names>K</given-names> </name><name name-style="western"><surname>Gomez</surname><given-names>H</given-names> </name><name name-style="western"><surname>Bethard</surname><given-names>S</given-names> </name></person-group><article-title>A survey of confidence estimation and calibration in large language models</article-title><source>Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics</source><publisher-name>Association for Computational Linguistics</publisher-name><fpage>6577</fpage><lpage>6595</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.naacl-long.366</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wei</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schuurmans</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Self-consistency improves chain of thought reasoning in language models</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 21, 2022</comment><pub-id pub-id-type="doi">10.48550/arXiv.2203.11171</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Manakul</surname><given-names>P</given-names> </name><name name-style="western"><surname>Liusie</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gales</surname><given-names>M</given-names> </name></person-group><article-title>SelfCheckGPT: zero-resource black-box hallucination detection for generative large language models</article-title><conf-name>Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Dec 6-10, 2023</conf-date><pub-id pub-id-type="doi">10.18653/v1/2023.emnlp-main.557</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cohen</surname><given-names>J</given-names> </name></person-group><article-title>Weighted kappa: nominal scale agreement with provision for scaled disagreement or partial credit</article-title><source>Psychol Bull</source><year>1968</year><month>10</month><volume>70</volume><issue>4</issue><fpage>213</fpage><lpage>220</lpage><pub-id pub-id-type="doi">10.1037/h0026256</pub-id><pub-id pub-id-type="medline">19673146</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Landis</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Koch</surname><given-names>GG</given-names> </name></person-group><article-title>The measurement of observer agreement for categorical data</article-title><source>Biometrics</source><year>1977</year><month>03</month><volume>33</volume><issue>1</issue><fpage>159</fpage><lpage>174</lpage><pub-id pub-id-type="doi">10.2307/2529310</pub-id><pub-id pub-id-type="medline">843571</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mandrekar</surname><given-names>JN</given-names> </name></person-group><article-title>Receiver operating characteristic curve in diagnostic test assessment</article-title><source>J Thorac Oncol</source><year>2010</year><month>09</month><volume>5</volume><issue>9</issue><fpage>1315</fpage><lpage>1316</lpage><pub-id pub-id-type="doi">10.1097/JTO.0b013e3181ec173d</pub-id><pub-id pub-id-type="medline">20736804</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ranganathan</surname><given-names>P</given-names> </name><name name-style="western"><surname>Pramesh</surname><given-names>CS</given-names> </name><name name-style="western"><surname>Buyse</surname><given-names>M</given-names> </name></person-group><article-title>Common pitfalls in statistical analysis: clinical versus statistical significance</article-title><source>Perspect Clin Res</source><year>2015</year><volume>6</volume><issue>3</issue><fpage>169</fpage><lpage>170</lpage><pub-id pub-id-type="doi">10.4103/2229-3485.159943</pub-id><pub-id pub-id-type="medline">26229754</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lyu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zapadka</surname><given-names>ME</given-names> </name><etal/></person-group><article-title>Translating radiology reports into plain language using ChatGPT and GPT-4 with prompt learning: results, limitations, and potential</article-title><source>Vis Comput Ind Biomed Art</source><year>2023</year><month>05</month><day>18</day><volume>6</volume><issue>1</issue><fpage>9</fpage><pub-id pub-id-type="doi">10.1186/s42492-023-00136-5</pub-id><pub-id pub-id-type="medline">37198498</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moor</surname><given-names>M</given-names> </name><name name-style="western"><surname>Banerjee</surname><given-names>O</given-names> </name><name name-style="western"><surname>Abad</surname><given-names>ZSH</given-names> </name><etal/></person-group><article-title>Foundation models for generalist medical artificial intelligence</article-title><source>Nature</source><year>2023</year><month>04</month><volume>616</volume><issue>7956</issue><fpage>259</fpage><lpage>265</lpage><pub-id pub-id-type="doi">10.1038/s41586-023-05881-4</pub-id><pub-id pub-id-type="medline">37045921</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Grani</surname><given-names>G</given-names> </name><name name-style="western"><surname>Sponziello</surname><given-names>M</given-names> </name><name name-style="western"><surname>Pecce</surname><given-names>V</given-names> </name><name name-style="western"><surname>Ramundo</surname><given-names>V</given-names> </name><name name-style="western"><surname>Durante</surname><given-names>C</given-names> </name></person-group><article-title>Contemporary thyroid nodule evaluation and management</article-title><source>J Clin Endocrinol Metab</source><year>2020</year><month>09</month><day>1</day><volume>105</volume><issue>9</issue><fpage>2869</fpage><lpage>2883</lpage><pub-id pub-id-type="doi">10.1210/clinem/dgaa322</pub-id><pub-id pub-id-type="medline">32491169</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hoang</surname><given-names>JK</given-names> </name><name name-style="western"><surname>Middleton</surname><given-names>WD</given-names> </name><name name-style="western"><surname>Farjat</surname><given-names>AE</given-names> </name><etal/></person-group><article-title>Reduction in thyroid nodule biopsies and improved accuracy with American College of Radiology Thyroid Imaging Reporting and Data System</article-title><source>Radiology</source><year>2018</year><month>04</month><volume>287</volume><issue>1</issue><fpage>185</fpage><lpage>193</lpage><pub-id pub-id-type="doi">10.1148/radiol.2018172572</pub-id><pub-id pub-id-type="medline">29498593</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mulherin</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>WC</given-names> </name></person-group><article-title>Spectrum bias or spectrum effect? Subgroup variation in diagnostic test evaluation</article-title><source>Ann Intern Med</source><year>2002</year><month>10</month><day>1</day><volume>137</volume><issue>7</issue><fpage>598</fpage><lpage>602</lpage><pub-id pub-id-type="doi">10.7326/0003-4819-137-7-200210010-00011</pub-id><pub-id pub-id-type="medline">12353947</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>A</given-names> </name><name name-style="western"><surname>Feng</surname><given-names>B</given-names> </name><name name-style="western"><surname>Xue</surname><given-names>B</given-names> </name><etal/></person-group><article-title>DeepSeek-v3 technical report</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 27, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2412.19437</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>A</given-names> </name><name name-style="western"><surname>Feng</surname><given-names>B</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>B</given-names> </name><etal/></person-group><article-title>DeepSeek-v2: a strong, economical, and efficient mixture-of-experts language model</article-title><source>arXiv</source><comment>Preprint posted online on  May 7, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2405.04434</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Shazeer</surname><given-names>N</given-names> </name><name name-style="western"><surname>Mirhoseini</surname><given-names>A</given-names> </name><name name-style="western"><surname>Maziarz</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Outrageously large neural networks: the sparsely-gated mixture-of-experts layer</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 23, 2017</comment><pub-id pub-id-type="doi">10.48550/arXiv.1701.06538</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>D</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><etal/></person-group><article-title>DeepSeek-R1 incentivizes reasoning in LLMs through reinforcement learning</article-title><source>Nature</source><year>2025</year><month>09</month><volume>645</volume><issue>8081</issue><fpage>633</fpage><lpage>638</lpage><pub-id pub-id-type="doi">10.1038/s41586-025-09422-z</pub-id><pub-id pub-id-type="medline">40962978</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Achiam</surname><given-names>J</given-names> </name><name name-style="western"><surname>Adler</surname><given-names>S</given-names> </name><name name-style="western"><surname>Agarwal</surname><given-names>S</given-names> </name></person-group><article-title>GPT-4 technical report</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 15, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2303.08774</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Hurst</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lerer</surname><given-names>A</given-names> </name><name name-style="western"><surname>Goucher</surname><given-names>AP</given-names> </name><etal/></person-group><article-title>GPT-4o system card</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 25, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2410.21276</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sim</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wright</surname><given-names>CC</given-names> </name></person-group><article-title>The kappa statistic in reliability studies: use, interpretation, and sample size requirements</article-title><source>Phys Ther</source><year>2005</year><month>03</month><volume>85</volume><issue>3</issue><fpage>257</fpage><lpage>268</lpage><pub-id pub-id-type="medline">15733050</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Feinstein</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Cicchetti</surname><given-names>DV</given-names> </name></person-group><article-title>High agreement but low kappa: I. The problems of two paradoxes</article-title><source>J Clin Epidemiol</source><year>1990</year><volume>43</volume><issue>6</issue><fpage>543</fpage><lpage>549</lpage><pub-id pub-id-type="doi">10.1016/0895-4356(90)90158-l</pub-id><pub-id pub-id-type="medline">2348207</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sorin</surname><given-names>V</given-names> </name><name name-style="western"><surname>Glicksberg</surname><given-names>BS</given-names> </name><name name-style="western"><surname>Artsi</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Utilizing large language models in breast cancer management: systematic review</article-title><source>J Cancer Res Clin Oncol</source><year>2024</year><month>03</month><day>19</day><volume>150</volume><issue>3</issue><fpage>140</fpage><pub-id pub-id-type="doi">10.1007/s00432-024-05678-6</pub-id><pub-id pub-id-type="medline">38504034</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pan</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Jiao</surname><given-names>FY</given-names> </name></person-group><article-title>DeepSeek perspective on managing Kawasaki disease in Chinese children</article-title><source>Zhongguo Dang Dai Er Ke Za Zhi</source><year>2025</year><month>05</month><day>15</day><volume>27</volume><issue>5</issue><fpage>524</fpage><lpage>528</lpage><pub-id pub-id-type="doi">10.7499/j.issn.1008-8830.2502042</pub-id><pub-id pub-id-type="medline">40462424</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Deng</surname><given-names>J</given-names> </name><name name-style="western"><surname>Qiu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Dong</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Evaluating ChatGPT and DeepSeek in postdural puncture headache management: a comparative study with international consensus guidelines</article-title><source>BMC Neurol</source><year>2025</year><month>07</month><day>1</day><volume>25</volume><issue>1</issue><fpage>264</fpage><pub-id pub-id-type="doi">10.1186/s12883-025-04280-8</pub-id><pub-id pub-id-type="medline">40597769</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Seth</surname><given-names>I</given-names> </name><name name-style="western"><surname>Marcaccini</surname><given-names>G</given-names> </name><name name-style="western"><surname>Lim</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Management of Dupuytren's disease: a multi-centric comparative analysis between experienced hand surgeons versus artificial intelligence</article-title><source>Diagnostics (Basel)</source><year>2025</year><month>02</month><day>28</day><volume>15</volume><issue>5</issue><fpage>587</fpage><pub-id pub-id-type="doi">10.3390/diagnostics15050587</pub-id><pub-id pub-id-type="medline">40075834</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Puladi</surname><given-names>B</given-names> </name><name name-style="western"><surname>Gsaxner</surname><given-names>C</given-names> </name><name name-style="western"><surname>Kleesiek</surname><given-names>J</given-names> </name><name name-style="western"><surname>H&#x00F6;lzle</surname><given-names>F</given-names> </name><name name-style="western"><surname>R&#x00F6;hrig</surname><given-names>R</given-names> </name><name name-style="western"><surname>Egger</surname><given-names>J</given-names> </name></person-group><article-title>The impact and opportunities of large language models like ChatGPT in oral and maxillofacial surgery: a narrative review</article-title><source>Int J Oral Maxillofac Surg</source><year>2024</year><month>01</month><volume>53</volume><issue>1</issue><fpage>78</fpage><lpage>88</lpage><pub-id pub-id-type="doi">10.1016/j.ijom.2023.09.005</pub-id><pub-id pub-id-type="medline">37798200</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lim</surname><given-names>B</given-names> </name><name name-style="western"><surname>Seth</surname><given-names>I</given-names> </name><name name-style="western"><surname>Kah</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Using generative artificial intelligence tools in cosmetic surgery: a study on rhinoplasty, facelifts, and blepharoplasty procedures</article-title><source>J Clin Med</source><year>2023</year><month>10</month><day>14</day><volume>12</volume><issue>20</issue><fpage>6524</fpage><pub-id pub-id-type="doi">10.3390/jcm12206524</pub-id><pub-id pub-id-type="medline">37892665</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Borna</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gomez-Cabello</surname><given-names>CA</given-names> </name><name name-style="western"><surname>Pressman</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Haider</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Forte</surname><given-names>AJ</given-names> </name></person-group><article-title>Comparative analysis of large language models in emergency plastic surgery decision-making: the role of physical exam data</article-title><source>J Pers Med</source><year>2024</year><month>06</month><day>8</day><volume>14</volume><issue>6</issue><fpage>612</fpage><pub-id pub-id-type="doi">10.3390/jpm14060612</pub-id><pub-id pub-id-type="medline">38929832</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jeblick</surname><given-names>K</given-names> </name><name name-style="western"><surname>Schachtner</surname><given-names>B</given-names> </name><name name-style="western"><surname>Dexl</surname><given-names>J</given-names> </name><etal/></person-group><article-title>ChatGPT makes medicine easy to swallow: an exploratory case study on simplified radiology reports</article-title><source>Eur Radiol</source><year>2024</year><month>05</month><volume>34</volume><issue>5</issue><fpage>2817</fpage><lpage>2825</lpage><pub-id pub-id-type="doi">10.1007/s00330-023-10213-1</pub-id><pub-id pub-id-type="medline">37794249</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cai</surname><given-names>LZ</given-names> </name><name name-style="western"><surname>Shaheen</surname><given-names>A</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Performance of generative large language models on ophthalmology board&#x2013;style questions</article-title><source>Am J Ophthalmol</source><year>2023</year><month>10</month><volume>254</volume><fpage>141</fpage><lpage>149</lpage><pub-id pub-id-type="doi">10.1016/j.ajo.2023.05.024</pub-id><pub-id pub-id-type="medline">37339728</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tangsrivimol</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Darzidehkalani</surname><given-names>E</given-names> </name><name name-style="western"><surname>Virk</surname><given-names>HUH</given-names> </name><etal/></person-group><article-title>Benefits, limits, and risks of ChatGPT in medicine</article-title><source>Front Artif Intell</source><year>2025</year><volume>8</volume><fpage>1518049</fpage><pub-id pub-id-type="doi">10.3389/frai.2025.1518049</pub-id><pub-id pub-id-type="medline">39949509</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bhayana</surname><given-names>R</given-names> </name><name name-style="western"><surname>Krishna</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bleakney</surname><given-names>RR</given-names> </name></person-group><article-title>Performance of ChatGPT on a radiology board-style examination: insights into current strengths and limitations</article-title><source>Radiology</source><year>2023</year><month>06</month><volume>307</volume><issue>5</issue><fpage>e230582</fpage><pub-id pub-id-type="doi">10.1148/radiol.230582</pub-id><pub-id pub-id-type="medline">37191485</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Luo</surname><given-names>PW</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Xie</surname><given-names>X</given-names> </name><etal/></person-group><article-title>DeepSeek vs ChatGPT: a comparison study of their performance in answering prostate cancer radiotherapy questions in multiple languages</article-title><source>Am J Clin Exp Urol</source><year>2025</year><volume>13</volume><issue>2</issue><fpage>176</fpage><lpage>185</lpage><pub-id pub-id-type="doi">10.62347/UIAP7979</pub-id><pub-id pub-id-type="medline">40400997</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Fu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>K</given-names> </name></person-group><article-title>Evaluating the performance of DeepSeek-R1 and DeepSeek-V3 versus OpenAI models in the Chinese National Medical Licensing Examination: cross-sectional comparative study</article-title><source>JMIR Med Educ</source><year>2025</year><month>11</month><day>14</day><volume>11</volume><fpage>e73469</fpage><pub-id pub-id-type="doi">10.2196/73469</pub-id><pub-id pub-id-type="medline">41237388</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Center-level methodological characteristics, cross-center variations in ultrasound reporting language, standardized prompts, operational definitions of nonstandard large language model outputs, and additional diagnostic performance analyses (pairwise comparisons, tie-breaking events, sensitivity analyses, and Bayesian prevalence-adjusted estimates).</p><media xlink:href="jmir_v28i1e93890_app1.docx" xlink:title="DOCX File, 39 KB"/></supplementary-material></app-group></back></article>