<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e94689</article-id><article-id pub-id-type="doi">10.2196/94689</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Error Detection and Correction in Chinese Radiology Reports Using Large Language Models: Real-World Clinical Validation Study</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Zhou</surname><given-names>Jiafeng</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Wei</surname><given-names>Yuxin</given-names></name><degrees>BME</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Cai</surname><given-names>Qian</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chen</surname><given-names>Yongchun</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Edzeafene-Mensah</surname><given-names>Eugene</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yang</surname><given-names>Yunjun</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Pan</surname><given-names>Zhifang</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref></contrib></contrib-group><aff id="aff1"><institution>Radiology Department, The First Affiliated Hospital of Wenzhou Medical University</institution><addr-line>Nanbaixiang, Ouhai District</addr-line><addr-line>Wenzhou</addr-line><addr-line>Zhejiang</addr-line><country>China</country></aff><aff id="aff2"><institution>The Eye Hospital of Wenzhou Medical University</institution><addr-line>Wenzhou</addr-line><country>China</country></aff><aff id="aff3"><institution>The First Affiliated Hospital of Wenzhou Medical University</institution><addr-line>Nanbaixiang, Ouhai District</addr-line><addr-line>Wenzhou</addr-line><addr-line>Zhejiang</addr-line><country>China</country></aff><aff id="aff4"><institution>School of Basic Medical Sciences, Wenzhou Medical University</institution><addr-line>Wenzhou</addr-line><addr-line>Zhejiang</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Guo</surname><given-names>Jinyu</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Wang</surname><given-names>Tianci</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Zhifang Pan, PhD, The First Affiliated Hospital of Wenzhou Medical University, Nanbaixiang, Ouhai District, Wenzhou, Zhejiang, China, 86 15088581221; <email>panzhifang@wmu.edu.cn</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>21</day><month>8</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e94689</elocation-id><history><date date-type="received"><day>07</day><month>03</month><year>2026</year></date><date date-type="rev-recd"><day>29</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>30</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Jiafeng Zhou, Yuxin Wei, Qian Cai, Yongchun Chen, Eugene Edzeafene-Mensah, Yunjun Yang, Zhifang Pan. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 21.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e94689"/><abstract><sec><title>Background</title><p>Large language models (LLMs) show promise in automatically detecting errors in radiology reports, but their performance remains insufficiently validated in large-scale, real-world clinical datasets.</p></sec><sec><title>Objective</title><p>This study aimed to systematically evaluate the performance of LLMs in detecting and correcting errors in Chinese radiology reports derived from authentic clinical data.</p></sec><sec sec-type="methods"><title>Methods</title><p>A large-scale dataset of 4480 Chinese radiology reports with modification records containing real clinical practice-generated errors was retrospectively collected between January 2023 and June 2024 at a single institution. After exclusions, 1363 reports containing 1551 errors were included. The dataset covers various anatomical parts of the body from different imaging modalities and was randomly divided into a test set (n=1263) and an internal validation set (n=100). Additionally, 100 error-free reports were added to the internal validation set. An additional 200 English-language reports from the Medical Information Mart for Intensive Care (MIMIC-III) were used for external validation. Eight human readers and 8 widely adopted LLMs, enhanced by prompt engineering, were tasked with error detection. Overall and subgroup detection performance and reading time were evaluated. Correction suggestions from the 2 best-performing LLMs were reviewed by a senior radiologist.</p></sec><sec sec-type="results"><title>Results</title><p>On the test set, DeepSeek-R1 achieved the highest overall detection rate at 89% (95% CI 87%-90%), significantly better than the other 7 models (<italic>P</italic>=.001-.007). On the internal validation set, DeepSeek-R1 and Claude-3.5-Sonnet achieved detection rates of 83% (100/120; 95% CI 76%-89%) and 80% (96/120; 95% CI 72%-86%), respectively. DeepSeek-R1 showed performance comparable to radiologists (83%, 95% CI 76%-89% vs 80%, 95% CI 72%-86% for junior radiologists and 78%, 95% CI 70%-85% for senior radiologists; <italic>P</italic>=.39 and <italic>P</italic>=.19, respectively) and significantly better performance than that of nonradiologists and nonphysicians (83%, 95% CI 76%-89% vs 66%, 95% CI 57%-74% and 38%, 95% CI 30%-47%; <italic>P</italic>&#x003C;.001, respectively). DeepSeek-R1 showed a false-positive rate comparable to radiologists (DeepSeek-R1 vs senior radiologists and junior radiologists, 3% vs 0% and 1%; <italic>P</italic>=.25 and <italic>P</italic>=.61, respectively) and a significantly lower rate than nonradiologists and nonphysicians (3% vs 13% and 17%; <italic>P</italic>=.02 and <italic>P</italic>=.002, respectively). On the external validation set, DeepSeek-R1 and Claude-3.5-Sonnet achieved detection rates of 94% (95% CI 89%-97%) and 93% (95% CI 88%-97%), respectively. The correction accuracy of DeepSeek-R1 and Claude-3.5-Sonnet was 95% and 91%, respectively.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Enhanced LLMs, particularly DeepSeek-R1, demonstrated robust performance in error detection and correction within real-world Chinese radiology reports, supporting their clinical use for automated quality assurance and integration into workflows to improve reporting accuracy and efficiency.</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>radiology reports</kwd><kwd>AI</kwd><kwd>error detection</kwd><kwd>quality control</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Radiology reports are critical for clinical decision-making, as they facilitate the communication of complex imaging findings in a comprehensible, accurate, and efficient manner, thereby guiding patient management and treatment [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. The importance of accurate radiology reports in medical practice cannot be overstated. Even minor errors in the reports have the potential to result in severe consequences, such as incorrect diagnoses or delayed treatments [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref4">4</xref>]. However, radiology reports are prone to errors due to unreliable speech recognition and cognitive fatigue resulting from increasing radiologist workloads and high-pressure clinical environments [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>].</p><p>Large language models (LLMs), as a new AI method, can learn complex language patterns and generate fluent and coherent text, demonstrating revolutionary potential in medical processing [<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref10">10</xref>]. GPT-4 has been demonstrated to have error detection performance comparable to board-certified radiologists and holds great promise for substantial reductions in work hours and costs [<xref ref-type="bibr" rid="ref11">11</xref>]. This indicates that LLMs can serve as an effective quality control tool in radiology reports [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>]. However, some critical gaps in current research hinder direct clinical application. First, most studies are based on single-language English datasets. Evidence indicates that GPT&#x2019;s performance in certain tasks differs between English and non-English environments due to differences in linguistic structures, medical terminology, and contextual nuances [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. Non-English radiology corpora&#x2014;especially Chinese, where logographic characters, abbreviated terminologies (eg, &#x201C;Ca&#x201D; for cancer), and hybrid Latin-Chinese descriptors abound&#x2014;pose additional lexical and semantic challenges that may degrade model performance [<xref ref-type="bibr" rid="ref16">16</xref>]. Second, many studies used artificially constructed and simulated errors. The generated data has low transparency, is difficult to verify, and may introduce new biases such as hallucinations, overfitting, and low generalization, which will hinder the local operation of LLMs in medical institutions [<xref ref-type="bibr" rid="ref2">2</xref>]. Third, prior work has focused almost exclusively on error detection; the capacity of LLMs to correct errors with clinically acceptable reasoning remains largely unexplored.</p><p>This study aimed to address these critical gaps by systematically evaluating the error detection and correction capabilities of prompt-enhanced LLMs using a large-scale Chinese radiology report dataset derived from real clinical workflows. We hypothesized that prompt-enhanced LLMs would have good performance in error detection and correction while maintaining high speed and acceptable false-positive rates (FPRs).</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Ethical Considerations</title><p>This study was approved by the institutional review board of The First Affiliated Hospital of Wenzhou Medical University (approval number KY2025-R084). The requirement for written informed consent was waived because of the retrospective design. All patient identifiers were removed before the reports were provided to the LLMs and human reviewers.</p></sec><sec id="s2-2"><title>Data Collection</title><p>A total of 4480 Chinese radiology reports with modification records were obtained from the radiology system of our hospital between January 2023 and June 2024. Reports needing image re-evaluation to verify errors, diagnoses revised due to misdiagnosis, and controversial reports were excluded, resulting in a final corpus of 1363 reports containing 1551 errors (X-ray: 27, computed tomography [CT]: 1250, and magnetic resonance imaging [MRI]: 86). The dataset covers various anatomical parts of the body from different imaging modalities. The errors in the reports were all generated by radiologists during routine clinical work. The dataset was randomly divided into a test set (n=1263) and an internal validation set (n=100). Additionally, 100 error-free reports were added to the internal validation set. For further validation of the generalizability of our approach, 200 English reports from the Medical Information Mart for Intensive Care (MIMIC-III) dataset were used, including 100 reports with 148 inserted errors and 100 error-free reports (see Note S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p></sec><sec id="s2-3"><title>Data Annotation</title><p>Errors in the reports were initially annotated by a junior radiologist and checked by basic LLMs (eg, DeepSeek-R1) to find potential errors (details are provided in Note S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The basic LLMs participated only as static annotation auxiliary tools, and their model parameters remained frozen throughout the entire process without any form of training or fine-tuning. All interactive data generated during the annotation phase were strictly limited to the inference level and did not affect model parameter updates through gradient feedback. LLMs served only as supplementary screening aids to flag potential candidate errors and did not determine the final reference standard. All annotations were subsequently reviewed by a senior radiologist who retained the authority to reject the suggestions of LLMs, thereby ensuring comprehensive error detection while minimizing the introduction of model bias. The report reviewed by the senior radiologist was the final annotated data. Errors were categorized into 5 types based on the characteristics of the Chinese language and previous studies [<xref ref-type="bibr" rid="ref11">11</xref>]: (1) omission, (2) addition, (3) semantic error, (4) location discrepancy, and (5) others. The detailed definition of each type of error is provided in Table S1 of <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. The severity of the errors was categorized as either &#x201C;clinically significant&#x201D; or &#x201C;not clinically significant&#x201D; in accordance with recent studies [<xref ref-type="bibr" rid="ref12">12</xref>]. Clinically significant errors were considered to be of such magnitude that they could alter the meaning of the report, thereby risking misinterpretation by the clinician.</p></sec><sec id="s2-4"><title>Prompt Engineering</title><p>To enhance the error detection ability of LLMs, we randomly selected 100 reports from the test set for multiple iterations of model optimization. First, the LLMs were assigned the role of a professional radiologist and informed that their core task was to detect errors in radiology reports. The detection scope was then restricted using the 5 error types and enhanced by integrating chain-of-thought reasoning [<xref ref-type="bibr" rid="ref17">17</xref>] along with a few-shot examples. In addition to thinking step by step, the model was further instructed to &#x201C;split reports for sentence-by-sentence review&#x201D; and to verify anatomical sites and medical terminology based on different categories. A true positive was defined as when the LLMs flagged a text segment that corresponded to errors in the reference standard (details are provided in Note S3 of <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). To mitigate the &#x201C;hallucinations&#x201D; [<xref ref-type="bibr" rid="ref8">8</xref>] phenomenon caused by LLMs&#x2019; automatic correction of diagnostic content, a clear restriction&#x2014;&#x201C;prohibiting modification of the original report contents&#x201D;&#x2014;was implemented. By using these instructions and restrictions, the accuracy of the model&#x2019;s error recognition at the word level was effectively improved. The prompt templates were continuously adjusted and improved during the experimental exploration process until the model no longer showed significant improvement. The temperature was set to 0.3 based on a subset, ensuring stability and high accuracy. Our prompt template and parameters are provided in Note S4 of <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. All of the LLMs were run once per report, with calling processes using a &#x201C;single report, single conversation, no context accumulation&#x201D; approach. This design ensured that the knowledge representation ability of LLMs remained unchanged, fundamentally avoiding the potential impact of data leakage on subsequent test set evaluation results.</p></sec><sec id="s2-5"><title>Study Design</title><p>To validate the application of LLMs on real clinical data, 2 experiments were conducted (<xref ref-type="fig" rid="figure1">Figure 1</xref>). In part 1, 8 widely used LLMs were evaluated for error detection on the test set. These included international mainstream models such as GPT-4, GPT-4o (OpenAI), Gemini-1.5-Pro (Google), Claude-3.5-Sonnet (Anthropic), and models primarily trained on Chinese corpora such as DeepSeek-V3, DeepSeek-R1 (DeepSeek), Qwen-Plus (Alibaba Cloud), and GLM-4 (Zhiyuan AI). The model&#x2019;s output included the analysis, error fragment, and revision. The error detection rate was used to evaluate the performance of the models, and the time consumed was recorded from prompt submission to final response. To facilitate quality management of error detection, the LLMs&#x2019; ability to classify error types was also evaluated. A retrieval-augmented generation [<xref ref-type="bibr" rid="ref18">18</xref>] and in-context learning [<xref ref-type="bibr" rid="ref19">19</xref>] framework was applied, retrieving the top 3 most relevant examples per test sample to perform error-type classification based on predefined categories. This task serves as an indirect evaluation of the overall improvement in the model&#x2019;s semantic understanding and reasoning abilities resulting from the proposed optimization.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>The overall structure of the study using enhanced large language model (LLMs), including annotation for Chinese radiology reports containing errors, 8 enhanced LLMs screened on the test set, comparison of the top 2 models with human readers on the internal validation set, review and validation of the correction suggestions from the LLMs by an expert radiologist, and validation of an external validation set with 200 English reports from MIMIC-III. MIMIC-III: Medical Information Mart for Intensive Care III.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e94689_fig01.png"/></fig><p>In part 2, the top 2 LLMs were further analyzed on an internal validation set and an external validation set. Eight human readers from various backgrounds&#x2014;including 2 senior and junior radiologists, 2 nonradiologists, and 2 nonphysicians&#x2014;were compared with the LLMs in terms of error detection performance and time consumption on the internal validation set. The analysis and correction results provided by the LLMs were evaluated by an experienced radiologist. Detailed information about the human readers is provided in Note S5 of <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-6"><title>Statistical Analysis</title><p>Statistical analyses were performed using Python (version 3.8.19) and the pandas library (version 2.0.3, NumFOCUS). A paired test was used to assess differences between the LLMs and human readers. <italic>P</italic> values &#x003C;.05 were considered statistically significant. Additionally, 95% CIs were calculated using the Wilson method. The model&#x2019;s error detection performance was evaluated by detection rate, while its classification performance was assessed through precision, recall, and <italic>F</italic><sub>1</sub>-score. All statistical analyses were performed at the error level, with each error treated as an independent observation rather than at the report level.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Characteristics of the Test Set</title><p>The study flowchart is shown in <xref ref-type="fig" rid="figure2">Figure 2</xref>. The test set comprised 1263 reports containing 1431 errors, including 552 omission errors, 183 addition errors, 339 semantic errors, 202 location discrepancy errors, and 155 categorized as others. Clinically significant errors outnumbered not clinically significant errors (808 vs 623).</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Study flowchart of real-world Chinese radiology report selection, dataset splitting, and external English validation for LLM error detection evaluation. CT: computed tomography; LLM: large language model; MIMIC-III: Medical Information Mart for Intensive Care III; MRI: magnetic resonance imaging.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e94689_fig02.png"/></fig></sec><sec id="s3-2"><title>Enhanced LLMs Performance</title><sec id="s3-2-1"><title>Error Detection</title><p>On the test set, the models primarily trained on the Chinese corpora had a comparable average detection rate to the international mainstream models (78%, 95% CI 77%-79% vs 77%, 95% CI 76%-78%). DeepSeek-R1 achieved the highest overall detection rate at 89% (95% CI 87%-90%), significantly better than the other 7 models (<italic>P</italic>=.001-.007; <xref ref-type="table" rid="table1">Table 1</xref>). Within each model, the detection rate showed no notable difference between clinically significant and not clinically significant errors (<xref ref-type="fig" rid="figure3">Figure 3A</xref>).</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>The detection performance of large language models on the test set.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom" colspan="6">Detection rate, % (95% CI)</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">Omission</td><td align="left" valign="bottom">Addition</td><td align="left" valign="bottom">Semantic error</td><td align="left" valign="bottom">Location discrepancy</td><td align="left" valign="bottom">Others</td><td align="left" valign="bottom">Total</td></tr></thead><tbody><tr><td align="left" valign="top">GPT<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>-4</td><td align="left" valign="top">64 (60-68<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup>)</td><td align="left" valign="top">52 (45-60)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">56 (51-61)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">85 (80-89)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">81 (74-87)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">65 (63-68)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td></tr><tr><td align="left" valign="top">GPT-4o</td><td align="left" valign="top">80 (76-83)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">62 (55-69)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">66 (61-71)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">90 (85-93)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">84 (77-89)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">76 (74-78)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td></tr><tr><td align="left" valign="top">Gemini-Pro</td><td align="left" valign="top">84 (81-87)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">71 (64-77)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">75 (70-79)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">96 (92-98)</td><td align="left" valign="top">87 (81-91)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">82 (80-84)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td></tr><tr><td align="left" valign="top">DeepSeek-V3</td><td align="left" valign="top">72 (68-75)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">47 (40-54)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">54 (48-59)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">87 (82-91)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">74 (67-80)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">67 (64-69)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td></tr><tr><td align="left" valign="top">GLM<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup>4-Plus</td><td align="left" valign="top">84 (80-87)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">57 (50-64)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">65 (60-70)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">93 (88-95)</td><td align="left" valign="top">74 (67-80)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">76 (74-78)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td></tr><tr><td align="left" valign="top">Qwen-Plus</td><td align="left" valign="top">78 (75-82)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">69 (62-75)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">79 (74-83)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">92 (88-95)</td><td align="left" valign="top">81 (74-86)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">79 (77-81)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td></tr><tr><td align="left" valign="top">Claude-3.5-Sonnet</td><td align="left" valign="top">84 (81-87)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">78 (71-83)</td><td align="left" valign="top">81 (76-84)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">96 (92-98)</td><td align="left" valign="top">93 (88-96)</td><td align="left" valign="top">85 (83-87)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td></tr><tr><td align="left" valign="top">DeepSeek-R1</td><td align="left" valign="top">89 (86-91)</td><td align="left" valign="top">81 (75-86)</td><td align="left" valign="top">86 (82-89)</td><td align="left" valign="top">96 (92-98)</td><td align="left" valign="top">94 (89-96)</td><td align="left" valign="top">89 (87-90)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>GPT is developed by OpenAI.</p></fn><fn id="table1fn2"><p><sup>b</sup>Indicates <italic>P</italic>&#x003C;.05.</p></fn><fn id="table1fn3"><p><sup>c</sup>GLM: generative language model developed by Zhipu AI.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>(A) The bar chart presents the detection rates of clinically significant and not clinically significant errors by large language models (LLMs). (B) The bar chart displays the total processing time by LLMs.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e94689_fig03.png"/></fig></sec><sec id="s3-2-2"><title>Reading Time</title><p>Claude-3.5-Sonnet consumed the shortest time (2.8 h), while DeepSeek-R1 consumed the longest time (17.1 h) on the test set (<xref ref-type="fig" rid="figure3">Figure 3B</xref>).</p></sec><sec id="s3-2-3"><title>Error Types Classification</title><p>Overall, DeepSeek-R1, Claude-3.5-Sonnet, and Gemini-Pro were ranked among the top 3 in error classification, with <italic>F</italic><sub>1</sub>-scores ranging from 0.87 to 0.90. DeepSeek-R1, Claude-3.5-Sonnet, and Gemini-Pro also demonstrated strong capabilities in identifying the severity of errors, with <italic>F</italic><sub>1</sub>-scores ranging from 0.90 to 0.93 (Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p></sec></sec><sec id="s3-3"><title>The Generalization Ability of LLMs</title><p>On the internal validation set (<xref ref-type="table" rid="table2">Table 2</xref>), DeepSeek-R1 achieved a detection rate of 83% (100/120, 95% CI 76%-89%), and Claude-3.5-Sonnet achieved a detection rate of 80% (96/120, 95% CI 72%-86%). Except for semantic errors, where DeepSeek-R1 had significantly higher detection rates (25/27, 93% vs 20/27, 74%; <italic>P</italic>=.01), no statistically significant differences in detection rates were observed between DeepSeek-R1 and Claude-3.5-Sonnet in the overall and subgroup analyses (<italic>P</italic>=.27-.99; <xref ref-type="table" rid="table3">Tables 3</xref><xref ref-type="table" rid="table4"/>-<xref ref-type="table" rid="table5">5</xref>). DeepSeek-R1 had a comparable FPR with Claude-3.5-Sonnet (3% vs 4%; <italic>P</italic>&#x003E;.99). On the external validation set, DeepSeek-R1 achieved a detection rate of 94% (139/148, 95% CI 89%-97%), and Claude-3.5-Sonnet achieved a detection rate of 93% (137/148, 95% CI 88%-97%). DeepSeek-R1 had a lower FPR compared to Claude-3.5-Sonnet (5% vs 4%; <italic>P</italic>&#x003E;.99).</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Error detection rates of the top 2 large language models on internal validation and external validation sets.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Reader</td><td align="left" valign="bottom" colspan="6">Detection rate, n/N (%)</td><td align="left" valign="bottom">FPR<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">Omission</td><td align="left" valign="bottom">Addition</td><td align="left" valign="bottom">Semantic error</td><td align="left" valign="bottom">Location discrepancy</td><td align="left" valign="bottom">Others</td><td align="left" valign="bottom">Total</td><td align="left" valign="bottom"/></tr></thead><tbody><tr><td align="left" valign="top" colspan="8">Internal validation set</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Claude-3.5-Sonnet</td><td align="left" valign="top">23/32 (72)</td><td align="left" valign="top">12/19 (63)</td><td align="left" valign="top">20/27 (74)</td><td align="left" valign="top">28/29 (97)</td><td align="left" valign="top">13/13 (100)</td><td align="left" valign="top">96/120 (80)</td><td align="left" valign="top">4</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek-R1</td><td align="left" valign="top">25/32 (78)</td><td align="left" valign="top">10/19 (53)</td><td align="left" valign="top">25/27 (93)</td><td align="left" valign="top">28/29 (97)</td><td align="left" valign="top">12/13 (92)</td><td align="left" valign="top">100/120 (83)</td><td align="left" valign="top">3</td></tr><tr><td align="left" valign="top" colspan="8">External validation set</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Claude-3.5-Sonnet</td><td align="left" valign="top">11/15 (73)</td><td align="left" valign="top">18/19 (95)</td><td align="left" valign="top">31/35 (89)</td><td align="left" valign="top">40/41 (98)</td><td align="left" valign="top">37/38 (97)</td><td align="left" valign="top">137/148 (93)</td><td align="left" valign="top">4</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek-R1</td><td align="left" valign="top">10/15 (67)</td><td align="left" valign="top">16/19 (84)</td><td align="left" valign="top">35/35 (100)</td><td align="left" valign="top">40/41 (98)</td><td align="left" valign="top">38/38 (100)</td><td align="left" valign="top">139/148 (94)</td><td align="left" valign="top">5</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>FPR: false-positive rate.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Comparison of error detection rates between large language models and human readers on the internal validation set.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Reader</td><td align="left" valign="bottom" colspan="2">Total</td><td align="left" valign="bottom" colspan="2">X-ray or MRI<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></td><td align="left" valign="bottom" colspan="2">CT<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">Detection rate, % (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td><td align="left" valign="bottom">Detection rate, % (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td><td align="left" valign="bottom">Detection rate, % (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Junior 1</td><td align="left" valign="top">78 (69-84)</td><td align="left" valign="top">.22</td><td align="left" valign="top">56 (34-75)</td><td align="left" valign="top">.02<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="top">81 (73-88)</td><td align="left" valign="top">.71</td></tr><tr><td align="left" valign="top">Junior 2</td><td align="left" valign="top">82 (74-88)</td><td align="left" valign="top">.73</td><td align="left" valign="top">67 (44-84)</td><td align="left" valign="top">.27</td><td align="left" valign="top">84 (76-90)</td><td align="left" valign="top">.84</td></tr><tr><td align="left" valign="top">Junior radiologists average</td><td align="left" valign="top">80 (72-86)</td><td align="left" valign="top">.39</td><td align="left" valign="top">61 (39-80)</td><td align="left" valign="top">.07</td><td align="left" valign="top">83 (74-89)</td><td align="left" valign="top">.92</td></tr><tr><td align="left" valign="top">Senior 1</td><td align="left" valign="top">76 (67-83)</td><td align="left" valign="top">.13</td><td align="left" valign="top">78 (55-91)</td><td align="left" valign="top">.67</td><td align="left" valign="top">75 (66-83)</td><td align="left" valign="top">.13</td></tr><tr><td align="left" valign="top">Senior 2</td><td align="left" valign="top">80 (72-86)</td><td align="left" valign="top">.47</td><td align="left" valign="top">61 (39-80)</td><td align="left" valign="top">.04<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="top">83 (75-89)</td><td align="left" valign="top">&#x003E;.99</td></tr><tr><td align="left" valign="top">Senior radiologists average</td><td align="left" valign="top">78 (70-85)</td><td align="left" valign="top">.19</td><td align="left" valign="top">69 (44-84)</td><td align="left" valign="top">.17</td><td align="left" valign="top">79 (71-86)</td><td align="left" valign="top">.39</td></tr><tr><td align="left" valign="top">Nonradiologist 1</td><td align="left" valign="top">65 (56-73)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="top">50 (29-71)</td><td align="left" valign="top">.06</td><td align="left" valign="top">68 (58-76)</td><td align="left" valign="top">.007<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td></tr><tr><td align="left" valign="top">Nonradiologist 2</td><td align="left" valign="top">67 (58-74)</td><td align="left" valign="top">.002<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="top">67 (44-84)</td><td align="left" valign="top">.19</td><td align="left" valign="top">67 (57-75)</td><td align="left" valign="top">.006<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td></tr><tr><td align="left" valign="top">Nonradiologists average</td><td align="left" valign="top">66 (57-74)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="top">58 (34-75)</td><td align="left" valign="top">.07</td><td align="left" valign="top">67 (57-75)</td><td align="left" valign="top">.002<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td></tr><tr><td align="left" valign="top">Nonphysician 1</td><td align="left" valign="top">46 (37-55)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="top">28 (12-51)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="top">49 (40-59)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td></tr><tr><td align="left" valign="top">Nonphysician 2</td><td align="left" valign="top">30 (23-39)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="top">17 (6-39)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="top">32 (24-42)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td></tr><tr><td align="left" valign="top">Nonphysicians average</td><td align="left" valign="top">38 (30-47)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="top">22 (9-45)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="top">41 (32-51)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td></tr><tr><td align="left" valign="top">Claude-3.5-Sonnet</td><td align="left" valign="top">80 (72-86)</td><td align="left" valign="top">.42</td><td align="left" valign="top">78 (55-91)</td><td align="left" valign="top">.67</td><td align="left" valign="top">80 (72-87)</td><td align="left" valign="top">.49</td></tr><tr><td align="left" valign="top">DeepSeek-R1</td><td align="left" valign="top">83 (76-89)</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top">83 (61-94)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">83 (75-89)</td><td align="left" valign="top">&#x2014;</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>MRI: magnetic resonance imaging.</p></fn><fn id="table3fn2"><p><sup>b</sup>CT: computed tomography.</p></fn><fn id="table3fn3"><p><sup>c</sup>Indicates <italic>P</italic>&#x003C;.05.</p></fn><fn id="table3fn4"><p><sup>d</sup>Not applicable.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Comparison of detection rates for different error types in radiology reports on the internal validation set.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Reader</td><td align="left" valign="bottom" colspan="2">Omission</td><td align="left" valign="bottom" colspan="2">Addition</td><td align="left" valign="bottom" colspan="2">Semantic error</td><td align="left" valign="bottom" colspan="2">Location discrepancy</td><td align="left" valign="bottom" colspan="2">Others</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">Detection rate, % (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td><td align="left" valign="bottom">Detection rate, % (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td><td align="left" valign="bottom">Detection rate, % (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td><td align="left" valign="bottom">Detection rate, % (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td><td align="left" valign="bottom">Detection rate, % (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Junior 1</td><td align="left" valign="top">56 (39-72)</td><td align="left" valign="top">.08</td><td align="left" valign="top">89 (69-97)</td><td align="left" valign="top">.03<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">78 (59-89)</td><td align="left" valign="top">.01<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">97 (83-99)</td><td align="left" valign="top">.33</td><td align="left" valign="top">69 (42-87)</td><td align="left" valign="top">.08</td></tr><tr><td align="left" valign="top">Junior 2</td><td align="left" valign="top">88 (72-95)</td><td align="left" valign="top">.08</td><td align="left" valign="top">79 (57-91)</td><td align="left" valign="top">.33</td><td align="left" valign="top">89 (72-96)</td><td align="left" valign="top">.08</td><td align="left" valign="top">79 (62-90)</td><td align="left" valign="top">.33</td><td align="left" valign="top">62 (36-82)</td><td align="left" valign="top">.04<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td></tr><tr><td align="left" valign="top">Junior radiologists average</td><td align="left" valign="top">72 (55-84)</td><td align="left" valign="top">.74</td><td align="left" valign="top">84 (62-94)</td><td align="left" valign="top">.047<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">83 (63-92)</td><td align="left" valign="top">.02<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">88 (74-96)</td><td align="left" valign="top">.82</td><td align="left" valign="top">65 (36-82)</td><td align="left" valign="top">.047<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td></tr><tr><td align="left" valign="top">Senior 1</td><td align="left" valign="top">81 (65-91)</td><td align="left" valign="top">.45</td><td align="left" valign="top">53 (32-73)</td><td align="left" valign="top">.08</td><td align="left" valign="top">70 (52-84)</td><td align="left" valign="top">.006<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">90 (74-96)</td><td align="left" valign="top">&#x003E;.99</td><td align="left" valign="top">77 (50-92)</td><td align="left" valign="top">.17</td></tr><tr><td align="left" valign="top">Senior 2</td><td align="left" valign="top">66 (48-80)</td><td align="left" valign="top">.21</td><td align="left" valign="top">89 (69-97)</td><td align="left" valign="top">.03<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">78 (59-89)</td><td align="left" valign="top">.02<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">97 (83-99)</td><td align="left" valign="top">.33</td><td align="left" valign="top">69 (42-87)</td><td align="left" valign="top">.08</td></tr><tr><td align="left" valign="top">Senior radiologists average</td><td align="left" valign="top">75 (58-87)</td><td align="left" valign="top">.76</td><td align="left" valign="top">66 (41-81)</td><td align="left" valign="top">.80</td><td align="left" valign="top">74 (55-87)</td><td align="left" valign="top">.003<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">93 (78-98)</td><td align="left" valign="top">.65</td><td align="left" valign="top">77 (50-92)</td><td align="left" valign="top">.10</td></tr><tr><td align="left" valign="top">Nonradiologist 1</td><td align="left" valign="top">69 (51-82)</td><td align="left" valign="top">.45</td><td align="left" valign="top">47 (27-68)</td><td align="left" valign="top">.54</td><td align="left" valign="top">56 (37-72)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">90 (74-96)</td><td align="left" valign="top">&#x003E;.99</td><td align="left" valign="top">46 (23-71)</td><td align="left" valign="top">.008<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td></tr><tr><td align="left" valign="top">Nonradiologist 2</td><td align="left" valign="top">56 (39-72)</td><td align="left" valign="top">.14</td><td align="left" valign="top">63 (41-81)</td><td align="left" valign="top">.72</td><td align="left" valign="top">67 (48-81)</td><td align="left" valign="top">.001<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">86 (69-95)</td><td align="left" valign="top">.71</td><td align="left" valign="top">54 (29-77)</td><td align="left" valign="top">.02<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td></tr><tr><td align="left" valign="top">Nonradiologists average</td><td align="left" valign="top">59 (42-74)</td><td align="left" valign="top">.12</td><td align="left" valign="top">58 (36-77)</td><td align="left" valign="top">&#x003E;.99</td><td align="left" valign="top">59 (41-75)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">88 (74-96)</td><td align="left" valign="top">.81</td><td align="left" valign="top">58 (36-82)</td><td align="left" valign="top">.006<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td></tr><tr><td align="left" valign="top">Nonphysician 1</td><td align="left" valign="top">31 (18-49)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">32 (15-54)</td><td align="left" valign="top">.06</td><td align="left" valign="top">41 (25-59)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">72 (54-85)</td><td align="left" valign="top">.13</td><td align="left" valign="top">54 (29-77)</td><td align="left" valign="top">.02<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td></tr><tr><td align="left" valign="top">Nonphysician 2</td><td align="left" valign="top">6 (2-20)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">32 (15-54)</td><td align="left" valign="top">.03<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">56 (37-72)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">24 (12-42)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">46 (23-71)</td><td align="left" valign="top">.008<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td></tr><tr><td align="left" valign="top">Nonphysicians average</td><td align="left" valign="top">20 (9-35)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">32 (15-54)</td><td align="left" valign="top">.04<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">44 (28-63)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">48 (31-66)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">54 (29-77)</td><td align="left" valign="top">.006<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td></tr><tr><td align="left" valign="top">Claude-3.5-Sonnet</td><td align="left" valign="top">72 (55-84)</td><td align="left" valign="top">.33</td><td align="left" valign="top">63 (41-81)</td><td align="left" valign="top">.27</td><td align="left" valign="top">74 (55-87)</td><td align="left" valign="top">.01<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">97 (83-99)</td><td align="left" valign="top">.33</td><td align="left" valign="top">100 (77-100)</td><td align="left" valign="top">.34</td></tr><tr><td align="left" valign="top">DeepSeek-R1</td><td align="left" valign="top">78 (61-89)</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">53 (32-73)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">93 (77-98)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">97 (83-99)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">92 (67-99)</td><td align="left" valign="top">&#x2014;</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Indicates <italic>P</italic>&#x003C;.05.</p></fn><fn id="table4fn2"><p><sup>b</sup>Not applicable.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Comparison of detection rates for the severity of errors on the internal validation set.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Reader</td><td align="left" valign="bottom" colspan="2">Not clinically significant</td><td align="left" valign="bottom" colspan="2">Clinically significant</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">Detection rate, % (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td><td align="left" valign="bottom">Detection rate, % (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Junior radiologist 1</td><td align="left" valign="top">77 (61-88)</td><td align="left" valign="top">.06</td><td align="left" valign="top">78 (68-85)</td><td align="left" valign="top">.84</td></tr><tr><td align="left" valign="top">Junior radiologist 2</td><td align="left" valign="top">86 (71-94)</td><td align="left" valign="top">.02<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup></td><td align="left" valign="top">80 (70-87)</td><td align="left" valign="top">.57</td></tr><tr><td align="left" valign="top">Junior radiologist average</td><td align="left" valign="top">80 (64-90)</td><td align="left" valign="top">.08</td><td align="left" valign="top">79 (70-87)</td><td align="left" valign="top">.91</td></tr><tr><td align="left" valign="top">Senior radiologist 1</td><td align="left" valign="top">69 (52-81)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup></td><td align="left" valign="top">79 (69-86)</td><td align="left" valign="top">.84</td></tr><tr><td align="left" valign="top">Senior radiologist 2</td><td align="left" valign="top">77 (61-88)</td><td align="left" valign="top">.03<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup></td><td align="left" valign="top">81 (72-88)</td><td align="left" valign="top">.67</td></tr><tr><td align="left" valign="top">Senior radiologist average</td><td align="left" valign="top">71 (55-84)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup></td><td align="left" valign="top">81 (70-87)</td><td align="left" valign="top">.55</td></tr><tr><td align="left" valign="top">Nonradiology physician 1</td><td align="left" valign="top">51 (36-67)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup></td><td align="left" valign="top">71 (60-79)</td><td align="left" valign="top">.20</td></tr><tr><td align="left" valign="top">Nonradiology physician 2</td><td align="left" valign="top">57 (41-72)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup></td><td align="left" valign="top">71 (60-79)</td><td align="left" valign="top">.20</td></tr><tr><td align="left" valign="top">Nonradiology physician average</td><td align="left" valign="top">57 (41-72)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup></td><td align="left" valign="top">69 (59-78)</td><td align="left" valign="top">.10</td></tr><tr><td align="left" valign="top">Nonradiologists 1</td><td align="left" valign="top">54 (38-70)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup></td><td align="left" valign="top">42 (32-53)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup></td></tr><tr><td align="left" valign="top">Nonradiologists 2</td><td align="left" valign="top">49 (33-64)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup></td><td align="left" valign="top">22 (15-32)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup></td></tr><tr><td align="left" valign="top">Nonradiologists average</td><td align="left" valign="top">53 (36-67)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup></td><td align="left" valign="top">32 (23-42)</td><td align="left" valign="top">&#x003C;.001<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup></td></tr><tr><td align="left" valign="top">Claude-3.5-Sonnet</td><td align="left" valign="top">86 (71-94)</td><td align="left" valign="top">.66</td><td align="left" valign="top">78 (68-85)</td><td align="left" valign="top">.50</td></tr><tr><td align="left" valign="top">DeepSeek-R1</td><td align="left" valign="top">91 (78-97)</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup></td><td align="left" valign="top">80 (70-87)</td><td align="left" valign="top">&#x2014;</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>Indicates <italic>P</italic>&#x003C;.05.</p></fn><fn id="table5fn2"><p><sup>b</sup>Not applicable.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-4"><title>Comparison Between LLMs and Human Readers</title><p>In the overall analysis, there were no statistically significant differences in the average performance in detection rate between DeepSeek-R1 and radiologists (DeepSeek-R1, junior radiologists, and senior radiologists: 83%, 95% CI 76%-89% vs 80%, 95% CI 72%-86% and 78%, 95% CI 70%-85%; <italic>P</italic>=.39 and <italic>P</italic>=.19, respectively). DeepSeek-R1 had a significantly higher average detection rate than nonradiologists and nonphysicians (83%, 95% CI 76%-89% vs 66%, 95% CI 57%-74% and 38%, 95% CI 30%-47%; <italic>P</italic>&#x003C;.001, respectively; <xref ref-type="table" rid="table3">Table 3</xref>).</p><p>In the subgroup analysis of different imaging modalities, there were no statistically significant differences in the detection rates of X-ray or MRI and CT reports between DeepSeek-R1 and radiologists (<italic>P</italic>=.07-.99). The detection rate of DeepSeek-R1 was significantly higher than the average detection rate of nonradiologists average in CT reports (83%, 95% CI 75%&#x2010;89% vs 67%, 95% CI 57%&#x2010;75%; <italic>P</italic>=.002) and nonphysicians average in CT and X-ray or MRI reports (83%, 95% CI 75%&#x2010;89% vs 22%, 95% CI 9%&#x2010;45% and 41%, 95% CI 32%&#x2010;51%; <italic>P</italic>&#x003C;.001 and <italic>P</italic>&#x003C;.001, respectively; <xref ref-type="table" rid="table3">Table 3</xref>).</p><p>The performance of DeepSeek-R1 in detecting addition errors was significantly worse than that of the best-performing radiologist (53%, 95% CI 32%-73% vs 89%, 95% CI 69%-97%; <italic>P</italic>=.03; <xref ref-type="table" rid="table4">Table 4</xref>). The performance of DeepSeek-R1 was significantly higher than that of nonradiologists in detecting semantic errors (93%, 95% CI 77%-98% vs 59%, 95% CI 41%-75%; <italic>P</italic>&#x003C;.001) and significantly higher than that of nonphysicians in detecting omission, semantic, and location discrepancy errors (78%, 95% CI 61%-89% vs 20%, 95% CI 9%-35%; <italic>P</italic>=.01; 93%, 95% CI 77%-98% vs 49%, 95% CI 28%-63%; <italic>P</italic>&#x003C;.001; 97%, 95% CI 83%-99% vs 48%, 95% CI 31%-66%; <italic>P</italic>&#x003C;.001; <xref ref-type="table" rid="table4">Table 4</xref>). There were no statistically significant differences in detection rates in the analysis of error severity between DeepSeek-R1 and radiologists (<italic>P</italic>=.11-.99). The performance of DeepSeek-R1 was significantly higher than that of nonradiologists and nonphysicians in detecting not clinically significant errors (91%, 95% CI 78%-97% vs 57%, 95% CI 41%-72% and 53%, 95% CI 36%-67%; <italic>P</italic>&#x003C;.001 and <italic>P</italic>&#x003C;.001, respectively), and significantly higher than nonphysicians in detecting clinically significant errors (80%, 95% CI 70%-87% vs 32%, 95% CI 23%-42%; <italic>P</italic>&#x003C;.001; <xref ref-type="table" rid="table5">Table 5</xref>).</p><p>The FPR did not show statistically significant differences between DeepSeek-R1 and radiologists (DeepSeek-R1 vs senior radiologists and junior radiologists: 3% vs 0% and 1%; <italic>P</italic>=.25 and <italic>P</italic>=.61, respectively). DeepSeek-R1 demonstrated a significantly lower FPR than nonradiologists and nonphysicians (3% vs 13% and 17%; <italic>P</italic>=.02 and <italic>P</italic>=.002, respectively; <xref ref-type="fig" rid="figure4">Figure 4A</xref>).</p><p>The total reading time of DeepSeek-R1 was 4.62 hours, significantly longer than that of the slowest nonphysician (4.62 h vs 3.6 h; <italic>P</italic>=.01). The total reading time of Claude-3.5-Sonnet was 0.44 hours, significantly shorter than that of the fastest nonradiologists (0.44 h vs 1.56 h; <italic>P</italic>&#x003C;.001; <xref ref-type="fig" rid="figure4">Figure 4B</xref>).</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>(A) False-positive rate of different human readers and large language models (LLMs). (B) Total reading time of LLMs and human readers.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e94689_fig04.png"/></fig></sec><sec id="s3-5"><title>Error Reasoning and Revision Quality</title><p>The correction accuracy of DeepSeek-R1 and Claude-3.5-Sonnet was 95% and 91%, respectively (<xref ref-type="fig" rid="figure5">Figure 5</xref>). The correction accuracy of DeepSeek-R1 in location discrepancy and other errors, as well as the correction accuracy of Claude-3.5-Sonnet in omission and addition errors, were all 100%.</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Evaluation of the validity of modification suggestions provided by the model for detected errors in reports.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e94689_fig05.png"/></fig></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This study systematically evaluated the ability of mainstream LLMs to detect and correct errors in Chinese radiology reports using a large-scale, real clinical scenario dataset. Enhanced by domain-specific prompts, chain-of-thought reasoning, and strict output constraints, DeepSeek-R1 achieved detection rates of 89% (95% CI 87%-90%), 83% (95% CI 76%-89%), and 94% (95% CI 89%-97%) across test, internal, and external validation sets, respectively, matching radiologists and outperforming nonradiologists and nonphysicians. Moreover, DeepSeek-R1 and Claude-3.5-Sonnet also corrected 95% and 91% of flagged errors, respectively. These findings offer new perspectives on the localization and application of LLMs for automated quality control in digital health care settings.</p></sec><sec id="s4-2"><title>Comparison with Prior Work</title><p>Our study expanded the application of LLMs to non-English environments by using the largest error-labeled Chinese radiology report dataset (1363 reports, 1551 real errors). Our prompt engineering strategy, integrating chain-of-thought reasoning and strict output constraints, was crucial in mitigating hallucinations and focusing models on detection, thereby enhancing their reliability for clinical integration [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. We found that DeepSeek-R1 achieved an 89% (95% CI 87%-90%) detection rate on the test set and 83% (95% CI 76%-89%) on the internal validation, translating to a 25% to 30% absolute gain over the 52% zero-shot ceiling reported by Yan et al [<xref ref-type="bibr" rid="ref20">20</xref>] for Claude-3.5-Sonnet in Chinese ultrasound reports. Moreover, DeepSeek-R1 achieved a 94% (95% CI 89%-97%) detection rate on the English radiology reports dataset (MIMIC-III) without retraining, suggesting that the core optimization framework is transferable across languages, mitigating the single-language limitation. On the contrary, international mainstream models such as the GPT-4 series exhibited only moderate performance. This disparity underscores the critical impact of linguistic and domain-specific adaptation. Chinese radiology reports present features including word segmentation ambiguity, prevalent terminology abbreviations (eg, &#x201C;Ca&#x201D; for cancer), variable omission of quantifiers, and context-dependent semantic expressions. These unique challenges demand a deeply localized LLM solution. DeepSeek-R1, pretrained on high-quality Chinese corpora rich in medical literature, explicitly embeds domain-specific knowledge through adaptive training [<xref ref-type="bibr" rid="ref21">21</xref>]. This design stands in contrast to GPT-4, whose training data is predominantly English and lacks publicly disclosed optimization for Chinese medical text [<xref ref-type="bibr" rid="ref22">22</xref>]. The present study highlights that integrating domain-specific knowledge and prompt engineering can unlock robust error detection for non-English clinical narratives while remaining transferable to English data.</p><p>A significant advancement of this study was the use of a real-world dataset, as opposed to artificially constructed or synthetic data prevalent in prior research. Gertz et al [<xref ref-type="bibr" rid="ref11">11</xref>] intentionally inserted 150 errors from 5 common error categories into 100 reports and found that the performance of GPT-4 was comparable to that of radiologists. Sun et al [<xref ref-type="bibr" rid="ref4">4</xref>] highlighted the potential of fine-tuned LLMs to enhance error detection using LLM-generated synthetic datasets. Although such reports offer better control and avoid privacy concerns and the bias present in real-world data, they may generate new biases, such as overfitting and low generalizability [<xref ref-type="bibr" rid="ref2">2</xref>]. There may be poor performance in real life if the generated data does not represent a few common scenarios. Other limitations include the low transparency of generated datasets, the risk of perpetuating human-generated bias, and the difficulty of validating them [<xref ref-type="bibr" rid="ref23">23</xref>]. It should be pointed out that we used artificially inserted errors as the external validation set, which may also not fully represent the complexity and subtlety of real-world English radiology report errors. To bridge these gaps, our primary Chinese dataset captures 1551 naturally occurring errors extracted from the routine radiology workflow, ensuring that the evaluated performance is representative of genuine clinical scenarios, enhancing the practical applicability of our findings.</p><p>Compared with human experts, enhanced LLMs, particularly DeepSeek-R1, demonstrated strong error-detection capabilities in radiology reports, surpassing the performance of human readers and matching radiologists on certain tasks. In addition, DeepSeek-R1 performed well in error-type classification and maintained stable detection across imaging modalities with a low FPR. These findings align with prior studies [<xref ref-type="bibr" rid="ref20">20</xref>] and support their potential as assistive tools for quality assurance, especially amid growing clinical workloads. To better apply LLMs to clinical practice, in addition to considering detection rate, time consumption should also be taken into account. We found that DeepSeek-R1 achieved a 4% increase in detection rate compared to Claude-3.5-Sonnet but required 6 times the processing time. DeepSeek-R1 has 671 billion parameters, 61 layers of transformers embedded with multihead latent attention and mixture-of-experts layers, generating thousands to tens of thousands of internal reasoning tokens before producing the final answer, resulting in significantly longer output times [<xref ref-type="bibr" rid="ref21">21</xref>]. However, Claude-3.5-Sonnet prioritizes speed through architectural simplicity and implicit inference. On the contrary, when this study was conducted, DeepSeek-R1 was priced at US $0.55 per million input tokens, which was significantly lower than Claude-3.5-Sonnet, priced at US $3.00 per million input tokens. Thus, low-cost DeepSeek-R1 may be more suitable for real-world implementation decisions, such as offline quality control rather than real-time assistance. In addition, previous studies have shown that LLMs achieved a lower mean correction cost per report than the most cost-efficient radiologist [<xref ref-type="bibr" rid="ref11">11</xref>]. Future efforts should focus on optimizing the balance between accuracy and efficiency through model compression and fine-tuning.</p><p>Notably, DeepSeek-R1 offers transparency and can be deployed locally within institutional information technology environments at substantially lower costs than proprietary models [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref24">24</xref>]. For local LLM deployment, there are still many challenges that need to be addressed. The first challenge is hardware constraints (eg, graphical processing unit [GPU] memory, input/output [I/O] bottlenecks) and integration hurdles (eg, heterogeneous picture archiving and communication system [PACS] interfaces). We can address these through elastic compute scaling, model quantization, domestic hardware alternatives, tiered storage, containerized deployment, and standardized HL7/FHIR (Health Level 7 International/Fast Health Care Interoperability Resources) application programming interfaces (APIs). For data security, a &#x201C;data never leaves the hospital&#x201D; architecture can be enforced using on-premises GPU inference, SM4/AES-256 encryption, automatic deidentification, retrieval-augmented generation for continuous knowledge updates, and confidence-based filtering to mitigate hallucinations. On the personnel and organizational front, nonintrusive user interfaces (inline highlighting, sidebar summaries) and explainable outputs (eg, error-type justifications) foster radiologist trust. The LLM only suggests corrections and deliberately disables auto-correction; the radiologist retains full authority to accept, modify, or reject each suggestion. Figure S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> illustrates the proposed clinical workflow: (1) radiologist completes report; (2) LLM runs in background (local deployment); (3) if no error detected &#x2192; report finalized; (4) if error detected &#x2192; LLM highlights error and suggests correction; (5) radiologist reviews and then accepts, rejects, or modifies; and (6) final report signed. By integrating via a local API, it eliminates the need to transfer protected data or retrain models, ensuring both high performance and privacy compliance, which can be integrated into radiology reporting systems and effectively handle real-world medical data in clinical practice (Figure S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p><p>A critical consideration for the clinical deployment of any AI-assisted error detection tool is its FPR, as excessive false alarms can undermine trust, increase radiologist workload, and lead to &#x201C;alert fatigue&#x201D; [<xref ref-type="bibr" rid="ref20">20</xref>]. We found that DeepSeek-R1 achieved an FPR of 3% on internal validation, which was comparable to radiologists and significantly lower than nonradiologists and nonphysicians. This indicates that enhanced LLMs can maintain high specificity and minimize unnecessary interruptions for clinicians. However, certain errors, such as minor diagnostic omissions and retained normal templates, were still missed (see Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). This necessitates that the final adjudication remain with the radiologist, reinforcing a human-in-the-loop paradigm where the AI acts as a sensitive assistant, not an autonomous arbiter.</p><p>Explainability is critical in health care AI to foster trust and support clinical decision-making. Prior studies have largely focused on error detection without evaluating LLMs&#x2019; ability to correct errors, which limits their explainability [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref12">12</xref>]. Kim et al [<xref ref-type="bibr" rid="ref25">25</xref>] demonstrated that GPT-4 could effectively revise head CT reports. Similarly, we found that DeepSeek-R1 achieved 95% accuracy in proposed corrections. Our LLMs can provide an analysis process, reasons for correction, and suggestions for correction to increase clinical applicability. Radiologists can understand the entire correction process in order to make better decisions (accept, modify, or reject). We conducted a comprehensive qualitative analysis of all corrections and found that the LLMs were reasonable for most types of corrections, including simple typographic or grammatical, intermediate terminology or lateralization, and complex semantic rephrasing. However, there are still a small number of corrections that cannot be accepted, such as the use of professional terminology, general conclusions, specific writing habits, and so on. Representative examples of successful corrections and failed corrections were added in Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. The accuracy of correction can be continuously improved by providing real-time feedback to the LLMs on the modifications and rejections provided by radiologists, in order to achieve personalized service levels. These results indicate that LLMs can support an interactive learning environment for residents by identifying common errors and providing real-time feedback, thereby promoting continuous education [<xref ref-type="bibr" rid="ref11">11</xref>].</p></sec><sec id="s4-3"><title>Limitations</title><p>This study has several limitations. First, its retrospective design may introduce selection bias. Second, error annotation was restricted to the text level; detecting imaging-report discordance still requires visual input. Future integration of vision-language models could enable end-to-end quality assurance. Third, LLM hallucination remains a key concern, undermining clinical reliability and necessitating expert verification. Although illusions were mitigated to some extent via lower temperature and prompt constraints, fundamental solutions will require greater improvements in model architecture, knowledge injection, and training data control. Fourth, it should be noted that the external validation set (MIMIC-III) used artificially inserted errors, whereas our primary Chinese dataset contained real clinical errors. This difference may affect the generalizability of the cross-language comparison. Although the high detection rates on MIMIC-III (94% for DeepSeek-R1) suggest that the prompt engineering framework is transferable across languages, the artificial nature of these errors may not fully represent the complexity and subtlety of real-world English radiology report errors. Therefore, our cross-language findings could be interpreted as preliminary evidence of generalizability. Future work should validate on real English clinical errors to confirm these observations. Finally, prompt optimization enhances performance but poses standardization challenges and may yield variable effects across models.</p></sec><sec id="s4-4"><title>Conclusions</title><p>Based on a large-scale and real-world clinical corpus, this study demonstrates that enhanced LLMs, particularly DeepSeek-R1, can effectively detect and correct errors in Chinese radiology reports. The findings validate the use of multilingual, clinically derived data and support the local deployment of open-source LLMs as transparent, cost-effective, and privacy-preserving tools within radiology workflows, thereby facilitating their transition toward clinical decision support. While our findings support the potential clinical use of LLMs for error detection, prospective multicenter validation and workflow-based safety assessment are required before routine implementation. In the future, we will conduct multicenter validation to assess generalizability and explore multimodal LLMs for verifying image-text consistency, advancing toward AI-driven quality control in practice.</p></sec></sec></body><back><ack><p>The authors would like to express their sincere gratitude to Professor Peiying Li and Chenchen Xu for their invaluable guidance and encouragement throughout this research.</p><p>The authors declare the use of generative AI (GAI) in the research and manuscript writing process. According to the Generative AI Delegation Taxonomy (2025), the following tasks were delegated to GAI tools under full human supervision: proofreading and editing. The GAI tool used was DeepSeek-R1. Responsibility for the final manuscript lies entirely with the authors. GAI tools are not listed as authors and do not bear responsibility for the final outcomes.</p></ack><notes><sec><title>Funding</title><p>This study was supported by research grants from the Natural Science Foundation of China (grant 82572173), the Joint Fund of Zhejiang Provincial Natural Science Foundation of China (grant LKLY25H180006), and Wenzhou Municipal Science and Technology Bureau (grant ZG2022015).</p></sec><sec><title>Data Availability</title><p>The datasets generated or analyzed during the study are available from the corresponding author upon reasonable request.</p></sec></notes><fn-group><fn fn-type="con"><p/><p>Conceptualization: JZ</p><p>Data curation: JZ, YW, QC, EE-M</p><p>Formal analysis: QC</p><p>Methodology: JZ</p><p>Project administration: ZP</p><p>Software: YW</p><p>Supervision: YY</p><p>Visualization: EE-M</p><p>Writing &#x2013; original draft: JZ, YW</p><p>Writing &#x2013; review &#x0026; editing: YC, ZP</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">API</term><def><p>application programming interface</p></def></def-item><def-item><term id="abb2">CT</term><def><p>computed tomography</p></def></def-item><def-item><term id="abb3">FPR</term><def><p>false-positive rate</p></def></def-item><def-item><term id="abb4">GPU</term><def><p>graphical processing unit</p></def></def-item><def-item><term id="abb5">HL7/FHIR</term><def><p>Health Level 7 International/Fast Health Care Interoperability Resources</p></def></def-item><def-item><term id="abb6">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb7">MIMIC-III</term><def><p>Medical Information Mart for Intensive Care III</p></def></def-item><def-item><term id="abb8">MRI</term><def><p>magnetic resonance imaging</p></def></def-item><def-item><term id="abb9">PACS</term><def><p>picture archiving and communication system</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mityul</surname><given-names>MI</given-names> </name><name name-style="western"><surname>Gilcrease-Garcia</surname><given-names>B</given-names> </name><name name-style="western"><surname>Mangano</surname><given-names>MD</given-names> </name><name name-style="western"><surname>Demertzis</surname><given-names>JL</given-names> </name><name name-style="western"><surname>Gunn</surname><given-names>AJ</given-names> </name></person-group><article-title>Radiology reporting: current practices and an introduction to patient-centered opportunities for improvement</article-title><source>AJR Am J Roentgenol</source><year>2018</year><month>02</month><volume>210</volume><issue>2</issue><fpage>376</fpage><lpage>385</lpage><pub-id pub-id-type="doi">10.2214/AJR.17.18721</pub-id><pub-id pub-id-type="medline">29140114</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Marrocchio</surname><given-names>C</given-names> </name><name name-style="western"><surname>Sverzellati</surname><given-names>N</given-names> </name></person-group><article-title>Will generative large language models become radiologists&#x2019; invaluable allies?</article-title><source>Radiology</source><year>2025</year><month>05</month><volume>315</volume><issue>2</issue><fpage>e251259</fpage><pub-id pub-id-type="doi">10.1148/radiol.251259</pub-id><pub-id pub-id-type="medline">40392094</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Minn</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Zandieh</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Filice</surname><given-names>RW</given-names> </name></person-group><article-title>Improving radiology report quality by rapidly notifying radiologist of report errors</article-title><source>J Digit Imaging</source><year>2015</year><month>08</month><volume>28</volume><issue>4</issue><fpage>492</fpage><lpage>498</lpage><pub-id pub-id-type="doi">10.1007/s10278-015-9781-9</pub-id><pub-id pub-id-type="medline">25694167</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sun</surname><given-names>C</given-names> </name><name name-style="western"><surname>Teichman</surname><given-names>K</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Generative large language models trained for detecting errors in radiology reports</article-title><source>Radiology</source><year>2025</year><month>05</month><volume>315</volume><issue>2</issue><fpage>e242575</fpage><pub-id pub-id-type="doi">10.1148/radiol.242575</pub-id><pub-id pub-id-type="medline">40392090</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Patel</surname><given-names>AG</given-names> </name><name name-style="western"><surname>Pizzitola</surname><given-names>VJ</given-names> </name><name name-style="western"><surname>Johnson</surname><given-names>CD</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>N</given-names> </name><name name-style="western"><surname>Patel</surname><given-names>MD</given-names> </name></person-group><article-title>Radiologists make more errors interpreting off-hours body CT studies during overnight assignments as compared with daytime assignments</article-title><source>Radiology</source><year>2020</year><month>11</month><volume>297</volume><issue>2</issue><fpage>374</fpage><lpage>379</lpage><pub-id pub-id-type="doi">10.1148/radiol.2020201558</pub-id><pub-id pub-id-type="medline">32808887</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alexander</surname><given-names>R</given-names> </name><name name-style="western"><surname>Waite</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bruno</surname><given-names>MA</given-names> </name><etal/></person-group><article-title>Mandating limits on workload, duty, and speed in radiology</article-title><source>Radiology</source><year>2022</year><month>08</month><volume>304</volume><issue>2</issue><fpage>274</fpage><lpage>282</lpage><pub-id pub-id-type="doi">10.1148/radiol.212631</pub-id><pub-id pub-id-type="medline">35699581</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sukhwal</surname><given-names>PC</given-names> </name><name name-style="western"><surname>Rajan</surname><given-names>V</given-names> </name><name name-style="western"><surname>Kankanhalli</surname><given-names>A</given-names> </name></person-group><article-title>A joint LLM-KG system for disease Q&#x0026;A</article-title><source>IEEE J Biomed Health Inform</source><year>2025</year><month>03</month><volume>29</volume><issue>3</issue><fpage>2257</fpage><lpage>2270</lpage><pub-id pub-id-type="doi">10.1109/JBHI.2024.3514659</pub-id><pub-id pub-id-type="medline">40030566</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bhayana</surname><given-names>R</given-names> </name></person-group><article-title>Chatbots and large language models in radiology: a practical primer for clinical and research applications</article-title><source>Radiology</source><year>2024</year><month>01</month><volume>310</volume><issue>1</issue><fpage>e232756</fpage><pub-id pub-id-type="doi">10.1148/radiol.232756</pub-id><pub-id pub-id-type="medline">38226883</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>TT</given-names> </name><name name-style="western"><surname>Makutonin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sirous</surname><given-names>R</given-names> </name><name name-style="western"><surname>Javan</surname><given-names>R</given-names> </name></person-group><article-title>Optimizing large language models in radiology and mitigating pitfalls: prompt engineering and fine-tuning</article-title><source>Radiographics</source><year>2025</year><month>04</month><volume>45</volume><issue>4</issue><fpage>e240073</fpage><pub-id pub-id-type="doi">10.1148/rg.240073</pub-id><pub-id pub-id-type="medline">40048389</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>T</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Experience-guided multi-agent interpretable framework for radiology report summarization</article-title><source>Comput Methods Programs Biomed</source><year>2026</year><month>01</month><volume>273</volume><fpage>109078</fpage><pub-id pub-id-type="doi">10.1016/j.cmpb.2025.109078</pub-id><pub-id pub-id-type="medline">41046706</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gertz</surname><given-names>RJ</given-names> </name><name name-style="western"><surname>Dratsch</surname><given-names>T</given-names> </name><name name-style="western"><surname>Bunck</surname><given-names>AC</given-names> </name><etal/></person-group><article-title>Potential of GPT-4 for detecting errors in radiology reports: implications for reporting accuracy</article-title><source>Radiology</source><year>2024</year><month>04</month><volume>311</volume><issue>1</issue><fpage>e232714</fpage><pub-id pub-id-type="doi">10.1148/radiol.232714</pub-id><pub-id pub-id-type="medline">38625012</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schmidt</surname><given-names>RA</given-names> </name><name name-style="western"><surname>Seah</surname><given-names>JCY</given-names> </name><name name-style="western"><surname>Cao</surname><given-names>K</given-names> </name><name name-style="western"><surname>Lim</surname><given-names>L</given-names> </name><name name-style="western"><surname>Lim</surname><given-names>W</given-names> </name><name name-style="western"><surname>Yeung</surname><given-names>J</given-names> </name></person-group><article-title>Generative large language models for detection of speech recognition errors in radiology reports</article-title><source>Radiol Artif Intell</source><year>2024</year><month>03</month><volume>6</volume><issue>2</issue><fpage>e230205</fpage><pub-id pub-id-type="doi">10.1148/ryai.230205</pub-id><pub-id pub-id-type="medline">38265301</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Salam</surname><given-names>B</given-names> </name><name name-style="western"><surname>St&#x00FC;we</surname><given-names>C</given-names> </name><name name-style="western"><surname>Nowak</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Large language models for error detection in radiology reports: a comparative analysis between closed-source and privacy-compliant open-source models</article-title><source>Eur Radiol</source><year>2025</year><month>08</month><volume>35</volume><issue>8</issue><fpage>4549</fpage><lpage>4557</lpage><pub-id pub-id-type="doi">10.1007/s00330-025-11438-y</pub-id><pub-id pub-id-type="medline">39979623</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cozzi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Pinker</surname><given-names>K</given-names> </name><name name-style="western"><surname>Hidber</surname><given-names>A</given-names> </name><etal/></person-group><article-title>BI-RADS category assignments by GPT-3.5, GPT-4, and Google Bard: a multilanguage study</article-title><source>Radiology</source><year>2024</year><month>04</month><volume>311</volume><issue>1</issue><fpage>e232133</fpage><pub-id pub-id-type="doi">10.1148/radiol.232133</pub-id><pub-id pub-id-type="medline">38687216</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Lai</surname><given-names>V</given-names> </name><name name-style="western"><surname>Ngo</surname><given-names>N</given-names> </name><name name-style="western"><surname>Pouran Ben Veyseh</surname><given-names>A</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Bouamor</surname><given-names>H</given-names> </name><name name-style="western"><surname>Pino</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bali</surname><given-names>K</given-names> </name></person-group><article-title>ChatGPT beyond English: towards a comprehensive evaluation of large language models in multilingual learning</article-title><source>Findings of the Association for Computational Linguistics</source><year>2023</year><fpage>13171</fpage><lpage>13189</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.findings-emnlp.878</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>T</given-names> </name><name name-style="western"><surname>Yue</surname><given-names>T</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>F</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>D</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Zong</surname><given-names>C</given-names> </name><name name-style="western"><surname>Xia</surname><given-names>F</given-names> </name><name name-style="western"><surname>Li</surname><given-names>W</given-names></name><name name-style="western"><surname>Navigli</surname><given-names>R</given-names> </name></person-group><article-title>PLOME: pre-training with misspelled knowledge for Chinese spelling correction</article-title><source>Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers)</source><year>2021</year><fpage>2991</fpage><lpage>3000</lpage><pub-id pub-id-type="doi">10.18653/v1/2021.acl-long.233</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tassone</surname><given-names>DM</given-names> </name><name name-style="western"><surname>Hitchcock</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Rossier</surname><given-names>CJ</given-names> </name><etal/></person-group><article-title>Evaluating chain-of-thought prompting in a GPT chatbot for BCID2 interpretation and stewardship: how does AI compare to human experts?</article-title><source>Antimicrob Steward Healthc Epidemiol</source><year>2025</year><volume>5</volume><issue>1</issue><fpage>e154</fpage><pub-id pub-id-type="doi">10.1017/ash.2025.10059</pub-id><pub-id pub-id-type="medline">40657035</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shi</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>T</given-names> </name><etal/></person-group><article-title>MKRAG: medical knowledge retrieval augmented generation for medical question answering</article-title><source>AMIA Annu Symp Proc</source><year>2025</year><volume>2024</volume><fpage>1011</fpage><lpage>1020</lpage><pub-id pub-id-type="medline">40417500</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lu</surname><given-names>YM</given-names> </name><name name-style="western"><surname>Letey</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zavatone-Veth</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Maiti</surname><given-names>A</given-names> </name><name name-style="western"><surname>Pehlevan</surname><given-names>C</given-names> </name></person-group><article-title>Asymptotic theory of in-context learning by linear attention</article-title><source>Proc Natl Acad Sci U S A</source><year>2025</year><month>07</month><day>15</day><volume>122</volume><issue>28</issue><fpage>e2502599122</fpage><pub-id pub-id-type="doi">10.1073/pnas.2502599122</pub-id><pub-id pub-id-type="medline">40632569</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yan</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>K</given-names> </name><name name-style="western"><surname>Feng</surname><given-names>B</given-names> </name><etal/></person-group><article-title>The use of large language models in detecting Chinese ultrasound report errors</article-title><source>NPJ Digit Med</source><year>2025</year><month>01</month><day>28</day><volume>8</volume><issue>1</issue><fpage>66</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01468-7</pub-id><pub-id pub-id-type="medline">39875800</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gibney</surname><given-names>E</given-names> </name></person-group><article-title>Scientists flock to DeepSeek: how they&#x2019;re using the blockbuster AI model</article-title><source>Nature</source><year>2025</year><month>01</month><day>29</day><pub-id pub-id-type="doi">10.1038/d41586-025-00275-0</pub-id><pub-id pub-id-type="medline">39881178</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Temperley</surname><given-names>HC</given-names> </name><name name-style="western"><surname>O&#x2019;Sullivan</surname><given-names>NJ</given-names> </name><name name-style="western"><surname>Mac Curtain</surname><given-names>BM</given-names> </name><etal/></person-group><article-title>Current applications and future potential of ChatGPT in radiology: a systematic review</article-title><source>J Med Imaging Radiat Oncol</source><year>2024</year><month>04</month><volume>68</volume><issue>3</issue><fpage>257</fpage><lpage>264</lpage><pub-id pub-id-type="doi">10.1111/1754-9485.13621</pub-id><pub-id pub-id-type="medline">38243605</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Smolyak</surname><given-names>D</given-names> </name><name name-style="western"><surname>Bjarnad&#x00F3;ttir</surname><given-names>MV</given-names> </name><name name-style="western"><surname>Crowley</surname><given-names>K</given-names> </name><name name-style="western"><surname>Agarwal</surname><given-names>R</given-names> </name></person-group><article-title>Large language models and synthetic health data: progress and prospects</article-title><source>JAMIA Open</source><year>2024</year><month>12</month><volume>7</volume><issue>4</issue><fpage>ooae114</fpage><pub-id pub-id-type="doi">10.1093/jamiaopen/ooae114</pub-id><pub-id pub-id-type="medline">39464796</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sandmann</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hegselmann</surname><given-names>S</given-names> </name><name name-style="western"><surname>Fujarski</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Benchmark evaluation of DeepSeek large language models in clinical decision-making</article-title><source>Nat Med</source><year>2025</year><month>08</month><volume>31</volume><issue>8</issue><fpage>2546</fpage><lpage>2549</lpage><pub-id pub-id-type="doi">10.1038/s41591-025-03727-2</pub-id><pub-id pub-id-type="medline">40267970</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>D</given-names> </name><name name-style="western"><surname>Shin</surname><given-names>HJ</given-names> </name><etal/></person-group><article-title>Large-scale validation of the feasibility of GPT-4 as a proofreading tool for head CT reports</article-title><source>Radiology</source><year>2025</year><month>01</month><volume>314</volume><issue>1</issue><fpage>e240701</fpage><pub-id pub-id-type="doi">10.1148/radiol.240701</pub-id><pub-id pub-id-type="medline">39873601</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Supporting notes, tables, and figures providing additional methodological details, evaluation results, and workflow illustrations.</p><media xlink:href="jmir_v28i1e94689_app1.docx" xlink:title="DOCX File, 172 KB"/></supplementary-material></app-group></back></article>