<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="review-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e97007</article-id><article-id pub-id-type="doi">10.2196/97007</article-id><article-categories><subj-group subj-group-type="heading"><subject>Review</subject></subj-group></article-categories><title-group><article-title>Effectiveness, Safety, and Workflow Burden of Large Language Model&#x2013;Based Medical Report Generation: Systematic Review</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Huang</surname><given-names>Jie-Lin</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhu</surname><given-names>Ji-Qing</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Ni</surname><given-names>Xiao-Guang</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1"/></contrib></contrib-group><aff id="aff1"><institution>Department of Endoscopy, National Cancer Center/National Clinical Research Center for Cancer/Cancer Hospital, Chinese Academy of Medical Sciences and Peking Union Medical College</institution><addr-line>No. 17 Panjiayuan South Lane, Chaoyang District</addr-line><addr-line>Beijing</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Brini</surname><given-names>Stefano</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Sisodiya</surname><given-names>Rajpal Singh</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Lu</surname><given-names>Yi</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Xiao-Guang Ni, MD, PhD, Department of Endoscopy, National Cancer Center/National Clinical Research Center for Cancer/Cancer Hospital, Chinese Academy of Medical Sciences and Peking Union Medical College, No. 17 Panjiayuan South Lane, Chaoyang District, Beijing, 100021, China, +86 10 8778 7606; <email>nixiaoguang@126.com</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>9</day><month>9</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e97007</elocation-id><history><date date-type="received"><day>02</day><month>04</month><year>2026</year></date><date date-type="rev-recd"><day>17</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>02</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Jie-Lin Huang, Ji-Qing Zhu, Xiao-Guang Ni. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 9.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e97007"/><abstract><sec><title>Background</title><p>Systems based on large language models (LLMs), multimodal LLMs, and vision-language foundation models are increasingly being evaluated for medical report generation in imaging and related clinical workflows. Existing reviews have summarized technical architectures, radiology applications, readability, and benchmark performance, but clinical readiness remains uncertain because safety, human oversight, and workflow outcomes are sparsely and inconsistently reported.</p></sec><sec><title>Objective</title><p>The aim of this study is to assess the effectiveness (expert acceptance and blinded preference), safety (clinically significant, omission, and commission errors), and workflow burden (reporting time, corrections, edit distance, and editing burden) of LLM-based medical report generation.</p></sec><sec sec-type="methods"><title>Methods</title><p>We searched PubMed/MEDLINE, Embase, Web of Science Core Collection, Scopus, and the Cochrane Library for studies published from January 1, 2016, through May 15, 2026. Eligible studies evaluated LLMs, multimodal LLMs, or vision-language foundation models for image-to-report generation, impression generation from findings, report drafting, or structured reporting in imaging workflows. Two reviewers performed screening, extraction, risk-of-bias assessment, and Grading of Recommendations Assessment, Development, and Evaluation&#x2013;informed narrative certainty assessment. Outcomes were clinically significant error rate, omission error rate, commission error rate, reporting time, edit burden, expert acceptance, and blinded expert preference. Meta-analysis was not performed because no comparable outcome had at least 2 studies with compatible task structure and analyzable data.</p></sec><sec sec-type="results"><title>Results</title><p>A total of 101 studies were included. Chest x-ray was the largest modality group (36 studies), followed by computed tomography, magnetic resonance imaging (MRI), ultrasound, endoscopy, pathology, ophthalmic, electrocardiographic, dental, and mixed-modality contexts. No study was judged at low risk of bias; 15 were moderate, 72 high, and 14 serious. Safety and workflow evidence remained heterogeneous and largely nonpoolable. In a chest x-ray study, AI report acceptance was similar to that of radiologist reports (6047/8580, 70.5% vs 6288/8580, 73.3%), but false-negative findings were slightly higher (1584/8580, 18.5% vs 1527/8580, 17.8%). In a clinician-collaboration chest x-ray study, AI reports were equivalent or preferred in 233 of 300 (77.7%) and 170 of 303 (56.1%) cases across 2 datasets; yet, clinically significant errors persisted. In a brain MRI study, AI assistance reduced reading time from 61 to 53 seconds, whereas impression drafting increased editing time and edit distance.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>This review shifts the synthesis from plausible report generation to clinically interpretable effectiveness, safety, and workflow effects. Expert acceptance and preference suggested assistive value in selected supervised settings, but these signals were limited by inconsistent reporting of clinically significant errors, omissions, commissions, and failed generations. Workflow effects were mixed, with some studies reporting shorter reading time or drafting support, and others reporting greater editing time or edit distance. The evidence remains too heterogeneous, biased, and sparse on case-level end points to support a pooled meta-analysis or autonomous clinical-readiness claims. Adoption should remain locally validated, clinician-supervised, and accompanied by standardized reporting of acceptance, preference, omissions, commissions, failed generations, reporting time, corrections, and editing burden.</p></sec><sec><title>Trial Registration</title><p>PROSPERO CRD420261302844; https://www.crd.york.ac.uk/PROSPERO/view/CRD420261302844</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>generative AI</kwd><kwd>medical report generation</kwd><kwd>radiology</kwd><kwd>pathology</kwd><kwd>endoscopy</kwd><kwd>workflow</kwd><kwd>systematic review</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Medical report generation is a central bottleneck in imaging practice. In radiology, pathology, and endoscopy, clinicians must convert findings into coherent, clinically actionable reports under rising volume, workforce pressure, fatigue, and time constraints. Clinical documentation consumes substantial physician time, and reporting work is a major part of the information-transfer burden in image-based specialties [<xref ref-type="bibr" rid="ref1">1</xref>]. Reports are also safety-critical communication artifacts: omissions, unsupported statements, ambiguous impressions, and delays in finalization can affect downstream clinical decisions. This reporting burden has therefore made automation and decision support an important target.</p><p>Earlier medical imaging report generation research predated the current large language model (LLM) wave and relied on neural architectures designed for specific text generation tasks rather than general purpose foundation models [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>]. More recent systems based on LLMs have been reported for radiology impression drafting and lumbar spine magnetic resonance imaging (MRI) reporting [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>], chest radiography and computed tomography (CT) report generation [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>], endoscopy reporting [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>], multimodal clinical or PACS (picture archiving and communication system)&#x2013;linked applications [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>], oral radiology and multimodal spinal MRI tasks [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>], and ultrasound reporting [<xref ref-type="bibr" rid="ref14">14</xref>]. These systems differ substantially in their inputs and intended roles: some generate complete reports from images, some draft impressions from existing findings, and others restructure or edit clinician-authored text. Treating these workflows as a single intervention can obscure important differences in risk, human oversight, and measurable benefit. Outcome interpretation, therefore, requires task-level classification rather than generic aggregation of all report-generation systems.</p><p>The central translational question remains unresolved: do these systems improve clinical reporting, or do they generate plausible but unsafe text? The literature is broad but fragmented across endoscopy [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>], mixed-modality or workflow-linked settings [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>], chest x-ray and cross-sectional imaging [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>], pathology [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>], and workflows ranging from autonomous generation to drafting, editing, and structuring support. This breadth does not yet translate into coherent clinical inference because studies differ in clinical input, generation task, comparator, human oversight, and outcome definition. A study that reports linguistic similarity to a reference report does not answer the same question as a study that measures missed findings, hallucinated statements, clinician editing burden, or final report acceptance.</p><p>The safety and workflow context for LLM-based report generation is broader than report text quality alone. Automation bias can reduce human verification of decision-support outputs when users overrely on automated suggestions, especially when verification is complex [<xref ref-type="bibr" rid="ref19">19</xref>]. Clinical AI safety literature also emphasizes that bias, distribution shift, unclear failure modes, and insufficient monitoring can create risks even when model performance appears favorable in development settings [<xref ref-type="bibr" rid="ref20">20</xref>]. For LLMs specifically, medical use raises concerns about fluent but unsupported outputs [<xref ref-type="bibr" rid="ref21">21</xref>], and hallucination is a recognized limitation of natural language generation systems [<xref ref-type="bibr" rid="ref22">22</xref>]. Recent radiology guidance on generative AI similarly emphasizes human oversight, local validation, monitoring, and workflow-specific safeguards before clinical deployment [<xref ref-type="bibr" rid="ref23">23</xref>]. These risks make safety and workflow end points central to assessing clinical usefulness rather than secondary technical details.</p><p>Existing reviews show why a new synthesis is needed now. Reviews and commentaries have described the rapid emergence of AI report generation, prompt engineering and fine-tuning, structured reporting, foundation models in imaging, and endoscopy applications [<xref ref-type="bibr" rid="ref24">24</xref>-<xref ref-type="bibr" rid="ref28">28</xref>]. More recent systematic and scoping reviews have summarized automated radiology report-generation methods, structured-reporting applications, and deep learning trends [<xref ref-type="bibr" rid="ref29">29</xref>-<xref ref-type="bibr" rid="ref31">31</xref>], while others have focused on LLM use in radiology reports, report readability or simplification, and broader radiology adoption trajectories [<xref ref-type="bibr" rid="ref32">32</xref>-<xref ref-type="bibr" rid="ref34">34</xref>]. These reviews are valuable, but generally emphasize technical progress, radiology-centered use cases, readability, or benchmark-oriented evaluation. They do not resolve whether the peer-reviewed evidence base reports on clinically interpretable safety and workflow outcomes consistently enough to guide supervised implementation or cumulative clinical inference. In particular, prior syntheses have not consistently separated autonomous image-to-report generation from supervised drafting or report-structuring support, although those categories carry different clinical responsibilities and failure modes.</p><p>This gap is important because technical evaluation and clinical decision-making require different evidence. Many studies still emphasize benchmark language metrics such as BLEU (Bilingual Evaluation Understudy), ROUGE (Recall-Oriented Understudy for Gisting Evaluation), CIDEr (Consensus-Based Image Description Evaluation), and BERTScore (a BERT-based semantic similarity metric, where BERT stands for Bidirectional Encoder Representations from Transformers) [<xref ref-type="bibr" rid="ref35">35</xref>-<xref ref-type="bibr" rid="ref38">38</xref>], while clinically important outcomes are reported less consistently. For this review, effectiveness was defined as expert acceptance and blinded expert preference; safety as clinically significant error rate, omission error rate, and commission error rate; and workflow burden as reporting time, number of corrections, edit distance, and editing burden. Here, omission errors are missed clinically relevant findings, and commission errors are unsupported or hallucinated findings introduced into the report. Benchmark similarity metrics help technical development but do not show whether a generated report missed a critical lesion, introduced a misleading statement, or reduced physicians&#x2019; finalization effort. Image-aware evaluation frameworks have begun to address this gap, but they remain uncommon and unharmonized for clinical synthesis [<xref ref-type="bibr" rid="ref39">39</xref>]. In addition, many reports provide reader ratings or preference judgments without case-level denominators, independent validation, or a clear account of failed generations. Consequently, individual positive studies do not yet establish whether the field is approaching clinical readiness or overstating progress through benchmark-dominated evaluation.</p><p>The objective of this systematic review was to assess the effectiveness (expert acceptance and blinded preference), safety (clinically significant errors, omission errors, and commission errors), and workflow burden (reporting time, corrections, edit distance, and editing burden) of LLM-based medical report generation across imaging modalities and related clinical reporting contexts. To support clinically meaningful interpretation, we separated task types and levels of human oversight so that autonomous image-to-report generation, supervised drafting, impression generation, and report-structuring support were not treated as equivalent interventions.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Protocol and Registration</title><p>This systematic review was conducted in accordance with the PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) 2020 statement (<xref ref-type="supplementary-material" rid="app3">Checklist 1</xref>) [<xref ref-type="bibr" rid="ref40">40</xref>]. Literature search reporting followed the PRISMA-S (Preferred Reporting Items for Systematic Reviews and Meta-Analyses Literature Search Extension) [<xref ref-type="bibr" rid="ref41">41</xref>], and narrative synthesis was reported with attention to the Synthesis Without Meta-Analysis (SWiM) guidance [<xref ref-type="bibr" rid="ref42">42</xref>]. The review question, primary outcomes, and subgroup structure were specified a priori, and the review was registered with PROSPERO (International Prospective Register of Systematic Reviews) 2026 (CRD420261302844) before formal screening began. The prespecified review protocol and extraction backbone are available from the corresponding author upon reasonable request. The PRISMA-S checklist was completed item by item in the <xref ref-type="supplementary-material" rid="app3">Checklist 1</xref>, and when a PRISMA-S item was not part of the search methodology or was not applicable, it was explicitly marked and justified.</p></sec><sec id="s2-2"><title>Eligibility Criteria</title><p>Eligible studies evaluated generative AI systems based on LLMs for generating an imaging report, report draft, or diagnostic impression from medical imaging, pathology whole-slide imaging, endoscopy, or multimodal, clinically relevant inputs. For eligibility, an intervention was considered eligible if an LLM, multimodal LLM, vision-language foundation model, or a language model component contributed directly to clinical report text generation, diagnostic impression generation, or clinical report drafting. Image classifiers, segmentation tools, retrieval systems, and image captioning systems were not eligible unless a language model or foundation model contributed directly to report generation.</p><p>Prespecified task types were report generation from images, generation of impressions from findings, and report drafting. Human oversight was categorized as autonomous generation, human-supervised drafting, or report editing or structuring support. Autonomous generation referred to systems that generated reports or report outputs from images or clinical inputs before human evaluation. Human-supervised drafting referred to workflows in which clinicians actively used, or iteratively collaborated on, AI-generated text during drafting. Report editing or structuring support referred to settings in which AI was applied after source report text, findings, or draft content was already available. We included randomized and nonrandomized comparative studies, reader studies, prospective workflow studies, retrospective validation studies, and single-arm technical validation studies if they used an explicit expert or clinical reference standard. Conference abstracts, reviews, editorials, letters without original data, dissertations, protocols, and preprints were excluded. Because preprints were excluded, this review was designed to evaluate peer-reviewed clinical evidence rather than the absolute frontier of technical model capability.</p><p>Effectiveness outcomes were expert acceptance rate and blinded expert preference. Safety outcomes were clinically significant error rate, omission error rate, and commission error rate. Workflow burden outcomes were reporting time, number of corrections, edit distance, and editing burden after generation. Omission errors were defined as missed clinically relevant findings, whereas commission errors were defined as unsupported or hallucinated findings introduced into the generated report. Exploratory outcomes included discrepancy rates between findings and impressions, and text similarity or language quality metrics, such as BLEU, ROUGE, CIDEr, and BERTScore.</p></sec><sec id="s2-3"><title>Information Sources</title><p>We searched PubMed/MEDLINE, Embase, Web of Science Core Collection, Scopus, and the Cochrane Library from January 1, 2016, through May 15, 2026. Europe PMC, OpenAlex, and reference lists of eligible studies and relevant reviews were used as supplemental sources through May 17, 2026, to identify additional records. Study registries were not used as bibliographic search sources because the eligibility criteria targeted completed peer-reviewed studies of medical report generation; this nonuse is reported explicitly in the PRISMA-S checklist. For studies that reported clinically important workflow or safety outcomes in nonconvertible formats, corresponding authors were contacted to request additional analyzable summary statistics or case-level denominator clarification.</p></sec><sec id="s2-4"><title>Search Strategy</title><p>The database strategy combined controlled vocabulary and free-text terms related to LLMs, multimodal LLMs, vision-language models, foundation models, and generative AI; report generation, report drafting, diagnostic impression generation, and structured reporting; and imaging domains including radiology, CT, MRI, radiography, pathology whole-slide imaging, ultrasound, endoscopy, and other medical image or signal interpretation settings. Exact database-specific search strings, interfaces, limits, dates, and record counts for PubMed/MEDLINE, Embase, Web of Science Core Collection, Scopus, and the Cochrane Library are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>; the completed PRISMA-S item-level reporting checklist is provided in the reporting guideline checklist file. Database searches were first completed on March 18, 2026, and were updated in May 2026 before final analysis. Across the original and updated database searches, 6936 records were identified before deduplication. Records retrieved in both searches were deduplicated by DOI, PMID, or normalized title and were counted once. After removal of 3008 duplicate records and 499 records already represented in the search corpus, 3429 database records remained for title and abstract screening.</p></sec><sec id="s2-5"><title>Study Selection</title><p>Title and abstract screening and full-text eligibility assessment were performed independently by 2 reviewers after deduplication, with disagreements resolved by discussion or adjudication by a third reviewer. No interrater reliability coefficient was generated; therefore, reviewer agreement is described procedurally rather than as a kappa statistic. Full-text exclusions were logged with one dominant reason per report.</p></sec><sec id="s2-6"><title>Data Extraction</title><p>Extraction frameworks at the study and outcome levels were used, and key extraction fields were checked in duplicate for studies that contributed clinically important safety or workflow outcomes. Extracted variables included study characteristics, clinical specialty, imaging modality, task type, level of human oversight, study design, validation setting, model characteristics, comparator type, reference standard, outcome definitions, and raw data sufficient for effect-size computation when available.</p></sec><sec id="s2-7"><title>Risk of Bias and Certainty Assessment</title><p>Risk of bias was assessed according to study design. Comparative and workflow studies were appraised using ROBINS-I (Risk of Bias in Non-Randomized Studies of Interventions) logic [<xref ref-type="bibr" rid="ref43">43</xref>], with reporting appraisal informed by CLAIM (Checklist for Artificial Intelligence in Medical Imaging) as an adjunct for assessing AI reporting quality [<xref ref-type="bibr" rid="ref44">44</xref>]. Retrospective technical validation studies that were not well captured by conventional intervention tools were assessed using a prespecified custom AI validation risk-of-bias framework. The core domains of that framework were representativeness of case selection, reference standard quality, blinding of expert assessment, validation independence, data leakage risk, model or prompt transparency, handling of failed or missing generations, and risk from outcome definition or selective reporting.</p><p>For the custom framework, moderate concern was assigned when a limitation was present but limited in scope or of limited consequence for clinical interpretation. High concern was assigned when a limitation was important and likely to influence interpretation, such as unclear validation independence, incomplete blinding, limited external validation, or incomplete failure reporting. Serious concern was assigned when a limitation could directly undermine clinical validity, such as likely data leakage, clearly nonrepresentative case selection, materially weak or unclear reference standards, the absence of handling of failed generations when failures could alter outcome interpretation, or strongly selective reporting of clinically important outcomes. The framework was used as a structured, qualitative, domain-based appraisal rather than a formally validated measurement instrument, and no interrater reliability coefficient was generated for this review. Its judgments should, therefore, be interpreted as transparent qualitative risk signals rather than as validated scale scores.</p><p>Because no outcomes met the criteria for meta-analysis, certainty was judged using a narrative certainty assessment informed by GRADE (Grading of Recommendations Assessment, Development, and Evaluation) domains rather than rating pooled effect estimates [<xref ref-type="bibr" rid="ref45">45</xref>]. We summarized these judgments in a GRADE Summary of Findings table format, using outcome-domain study counts from the selected clinically interpretable end point studies, and reported relative effects as not pooled or not estimable when compatible effect structures were unavailable.</p></sec><sec id="s2-8"><title>Data Synthesis</title><p>Eligibility for quantitative synthesis was assessed at the outcome level according to clinical coherence, task structure, comparator or reference standard design, and availability of analyzable event counts or summary statistics. For outcomes amenable to direct quantitative reconstruction, planned effect measures were risk differences for dichotomous outcomes and mean differences for continuous outcomes, with 95% CIs when derivable from published data. Outcomes reported across reader-case pairs were not treated as independent case-level events when clustering could not be reconstructed.</p><p>Quantitative synthesis was prespecified only for outcomes reported by at least 2 studies with compatible definitions, task structures, comparator or reference standard designs, and analyzable data structures. Outcomes that did not meet these criteria were summarized using structured narrative synthesis. When individual studies provided sufficient data, targeted quantitative reconstruction was performed to support interpretation without pooling. Narrative synthesis was organized by modality, task type, outcome domain, and level of human oversight. Formal subgroup, sensitivity, heterogeneity, and reporting bias analyses were reserved for poolable synthesis sets.</p></sec><sec id="s2-9"><title>Deviations From Protocol</title><p>The protocol anticipated quantitative synthesis when clinically and methodologically compatible synthesis sets were available. Supplemental citation discovery and reference screening were added to improve retrieval completeness. These procedures did not change the review question, primary outcomes, eligibility framework, or planned subgroup structure. When outcomes did not meet pooling criteria, synthesis followed the prespecified nonpooled approach of structured narrative synthesis with targeted quantitative reconstruction of individual studies where defensible.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Study Selection</title><p>The full study selection process is shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>. Across the original and updated database searches, 6936 records were identified before deduplication. After removal of 3008 duplicate records and 499 records already represented in the search corpus, 3429 database records proceeded to title and abstract screening. Public supplemental screening and citation-discovery screening contributed 1 additional eligible study, whereas backward reference screening did not identify additional reports for full-text assessment. Overall, 101 studies met the inclusion criteria for this systematic review. Study-level records are provided in Supplementary Table S1 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>, and full-text exclusions are listed in Supplementary Table S2 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) 2020 flow diagram for the systematic review. The figure summarizes database searching through May 15, 2026, other-method identification through supplemental citation discovery and backward reference screening, full-text assessment, and final inclusion of 101 studies. LLM: large language model; VLM: vision-language model.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e97007_fig01.png"/></fig></sec><sec id="s3-2"><title>Study Characteristics</title><p>The included literature was broad in technical scope but remained concentrated in a limited set of clinical scenarios (<xref ref-type="table" rid="table1">Table 1</xref>). Of the 101 included studies, 68 (67.3%) were retrospective or technical validation studies, 28 (27.7%) were nonrandomized comparative, reader, or workflow studies, and 5 (5.0%) were prospective studies or real-world workflow or validation studies. Chest x-ray remained the largest modality group (36, 35.6%), followed by mixed-modality settings (20, 19.8%), CT, CT angiography, or positron emission tomography/CT (11, 10.9%), MRI (10, 9.9%), ultrasound (6, 5.9%), endoscopy (5, 5.0%), pathology reporting settings (5, 5.0%), other x-ray or radiography (5, 5.0%), ophthalmic imaging (2, 2.0%), and electrocardiogram (1, 1.0%). By task type, 61 (60.4%) studies evaluated report generation from images, 29 (28.7%) evaluated report drafting or structured report generation, and 11 (10.9%) evaluated impression or diagnostic conclusion generation from findings or reports. By human oversight, 66 (65.3%) studies were classified as autonomous generation, 15 (14.9%) as human-supervised drafting, and 20 (19.8%) as report editing or structuring support. The relationship among modality, task type, and human oversight is shown in <xref ref-type="fig" rid="figure2">Figure 2</xref>, and the rapid expansion of the literature after 2023 is summarized in <xref ref-type="fig" rid="figure3">Figure 3A</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Evidence profile of included studies.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Domain and category</td><td align="left" valign="bottom">Studies, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Overall</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Total included studies</td><td align="left" valign="top">101</td></tr><tr><td align="left" valign="top" colspan="2">Study design</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Retrospective or technical validation</td><td align="left" valign="top">68 (67.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Nonrandomized comparative, reader, or workflow study</td><td align="left" valign="top">28 (27.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Prospective or real-world workflow or validation study</td><td align="left" valign="top">5 (5.0)</td></tr><tr><td align="left" valign="top" colspan="2">Clinical source</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Chest x-ray</td><td align="left" valign="top">36 (35.6)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mixed modality</td><td align="left" valign="top">20 (19.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>CT<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>, CTA<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup>, or PET/CT<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">11 (10.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MRI<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td><td align="left" valign="top">10 (9.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ultrasound</td><td align="left" valign="top">6 (5.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Endoscopy</td><td align="left" valign="top">5 (5.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Pathology report or whole-slide image</td><td align="left" valign="top">5 (5.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Other x-ray or radiography</td><td align="left" valign="top">5 (5.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ophthalmic imaging</td><td align="left" valign="top">2 (2.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Electrocardiogram</td><td align="left" valign="top">1 (1.0)</td></tr><tr><td align="left" valign="top" colspan="2">Generation task</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Image to report generation</td><td align="left" valign="top">61 (60.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Report drafting or structuring</td><td align="left" valign="top">29 (28.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Findings or report to impression generation</td><td align="left" valign="top">11 (10.9)</td></tr><tr><td align="left" valign="top" colspan="2">AI role in reporting workflow</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Autonomous generation</td><td align="left" valign="top">66 (65.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Human-supervised drafting</td><td align="left" valign="top">15 (14.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Report editing or structuring support</td><td align="left" valign="top">20 (19.8)</td></tr><tr><td align="left" valign="top" colspan="2">Overall risk of bias</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Low</td><td align="left" valign="top">0 (0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Moderate</td><td align="left" valign="top">15 (14.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>High</td><td align="left" valign="top">72 (71.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Serious</td><td align="left" valign="top">14 (13.9)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>CT: computed tomography.</p></fn><fn id="table1fn2"><p><sup>b</sup>CTA: computed tomography angiography.</p></fn><fn id="table1fn3"><p><sup>c</sup>PET: positron emission tomography.</p></fn><fn id="table1fn4"><p><sup>d</sup>MRI: magnetic resonance imaging.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Sankey diagram linking clinical modality or report source, report generation task, and AI role in the reporting workflow for the 101-study evidence base. Flow width is proportional to the number of studies in each pathway. The figure shows continued concentration in x-ray or radiography, report generation from images, and autonomous generation, while also showing smaller evidence streams in computed tomography (CT), computed tomography angiography (CTA), or positron emission tomography (PET)/CT; magnetic resonance imaging (MRI); ultrasound; endoscopy; pathology; ophthalmic imaging or electrocardiogram (ECG); mixed modality; human-supervised drafting; and report editing or structuring workflows.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e97007_fig02.png"/></fig><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Temporal evolution, evaluation signals, and domain-level risk of bias. Panel A summarizes annual counts among studies for which a publication year was captured, stratified by broad model family or by multimodal or hybrid grouping. Panel B displays nonexclusive standardized evaluation signal categories across the 101-study evidence base; individual automated metric names were grouped under a single automated metrics category because studies used heterogeneous benchmark measures. Panel C summarizes domain-level risk-of-bias judgments across custom AI validation domains (n=68) and ROBINS-I-informed domains (n=33). LLM: large language model; RAG: retrieval-augmented generation; ROBINS-I: Risk of Bias in Non-Randomized Studies of Interventions; VLM: vision-language model.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e97007_fig03.png"/></fig><p>Outcome reporting was imbalanced. Benchmark metrics such as BLEU, ROUGE, CIDEr, and BERTScore were common, especially in chest x-ray and technical validation studies, whereas clinically important outcomes, including significant errors, omissions, commissions, expert acceptance, blinded preference, reporting time, and edit burden, were inconsistently reported and used heterogeneous denominators. Thirty-six studies reported at least one clinically relevant human evaluation, safety, workflow, report quality, acceptability, or time outcome, but these outcomes differed by task, comparator, denominator, and reporting format. Benchmark-heavy chest x-ray studies were represented among the included studies [<xref ref-type="bibr" rid="ref46">46</xref>-<xref ref-type="bibr" rid="ref50">50</xref>]. Standardized evaluation signal reporting is summarized in <xref ref-type="fig" rid="figure3">Figure 3B</xref>. The subset of studies with clinically interpretable effectiveness, safety, and workflow-burden end points is summarized in <xref ref-type="table" rid="table2">Table 2</xref>.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Selected studies with clinically interpretable evaluation end points<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study and year</td><td align="left" valign="bottom">Clinical source</td><td align="left" valign="bottom">Generation task</td><td align="left" valign="bottom">AI role in reporting workflow</td><td align="left" valign="bottom">Assessment method</td><td align="left" valign="bottom">Reported evaluation outcomes</td></tr></thead><tbody><tr><td align="left" valign="top">Hong et al [<xref ref-type="bibr" rid="ref51">51</xref>], 2025</td><td align="left" valign="top">Chest x-ray</td><td align="left" valign="top">Image to report</td><td align="left" valign="top">Autonomous generation</td><td align="left" valign="top">Human reader assessment</td><td align="left" valign="top">Acceptance, safety</td></tr><tr><td align="left" valign="top">Tanno et al [<xref ref-type="bibr" rid="ref52">52</xref>], 2025</td><td align="left" valign="top">Chest x-ray</td><td align="left" valign="top">Image to report</td><td align="left" valign="top">Human-supervised drafting</td><td align="left" valign="top">Human reader assessment</td><td align="left" valign="top">Acceptance, safety</td></tr><tr><td align="left" valign="top">Wang et al [<xref ref-type="bibr" rid="ref53">53</xref>], 2026</td><td align="left" valign="top">MRI<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td><td align="left" valign="top">Findings to impression</td><td align="left" valign="top">Report editing or structuring support</td><td align="left" valign="top">Human reader assessment</td><td align="left" valign="top">Accuracy, efficiency</td></tr><tr><td align="left" valign="top">Serapio et al [<xref ref-type="bibr" rid="ref54">54</xref>], 2024</td><td align="left" valign="top">Mixed modality</td><td align="left" valign="top">Findings to impression</td><td align="left" valign="top">Human-supervised drafting</td><td align="left" valign="top">Workflow comparison</td><td align="left" valign="top">Efficiency</td></tr><tr><td align="left" valign="top">Li et al [<xref ref-type="bibr" rid="ref55">55</xref>], 2025</td><td align="left" valign="top">CT<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td><td align="left" valign="top">Image to report</td><td align="left" valign="top">Autonomous generation</td><td align="left" valign="top">Human expert assessment</td><td align="left" valign="top">Report quality, accuracy</td></tr><tr><td align="left" valign="top">Luo et al [<xref ref-type="bibr" rid="ref56">56</xref>], 2026</td><td align="left" valign="top">Endoscopy</td><td align="left" valign="top">Image to report</td><td align="left" valign="top">Autonomous generation</td><td align="left" valign="top">Human expert assessment</td><td align="left" valign="top">Report quality, accuracy</td></tr><tr><td align="left" valign="top">Ayaz et al [<xref ref-type="bibr" rid="ref57">57</xref>], 2025</td><td align="left" valign="top">Mixed modality</td><td align="left" valign="top">Image to report</td><td align="left" valign="top">Autonomous generation</td><td align="left" valign="top">Human expert assessment</td><td align="left" valign="top">Report quality</td></tr><tr><td align="left" valign="top">Li et al [<xref ref-type="bibr" rid="ref58">58</xref>], 2026</td><td align="left" valign="top">Mixed modality</td><td align="left" valign="top">Findings to impression</td><td align="left" valign="top">Human-supervised drafting</td><td align="left" valign="top">Clinical validation</td><td align="left" valign="top">Report quality, accuracy, efficiency</td></tr><tr><td align="left" valign="top">de Margerie-Mellon et al [<xref ref-type="bibr" rid="ref59">59</xref>], 2026</td><td align="left" valign="top">Mixed modality</td><td align="left" valign="top">Report drafting or structuring</td><td align="left" valign="top">Human-supervised drafting</td><td align="left" valign="top">Workflow comparison</td><td align="left" valign="top">Safety, efficiency</td></tr><tr><td align="left" valign="top">Jiang et al [<xref ref-type="bibr" rid="ref60">60</xref>], 2026</td><td align="left" valign="top">Endoscopy</td><td align="left" valign="top">Image to report</td><td align="left" valign="top">Autonomous generation</td><td align="left" valign="top">Clinical validation</td><td align="left" valign="top">Safety, accuracy, efficiency</td></tr><tr><td align="left" valign="top">Tekdemir et al [<xref ref-type="bibr" rid="ref61">61</xref>], 2026</td><td align="left" valign="top">CT</td><td align="left" valign="top">Findings to impression</td><td align="left" valign="top">Autonomous generation</td><td align="left" valign="top">Human expert assessment</td><td align="left" valign="top">Report quality, safety</td></tr><tr><td align="left" valign="top">Lorusso et al [<xref ref-type="bibr" rid="ref62">62</xref>], 2026</td><td align="left" valign="top">CTA<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="top">Findings to impression</td><td align="left" valign="top">Autonomous generation</td><td align="left" valign="top">Reference standard comparison</td><td align="left" valign="top">Accuracy, efficiency</td></tr><tr><td align="left" valign="top">Dong et al [<xref ref-type="bibr" rid="ref63">63</xref>], 2025</td><td align="left" valign="top">MRI</td><td align="left" valign="top">Report drafting or structuring</td><td align="left" valign="top">Human-supervised drafting</td><td align="left" valign="top">Human expert assessment</td><td align="left" valign="top">Report quality, accuracy, efficiency</td></tr><tr><td align="left" valign="top">Pellegrini et al [<xref ref-type="bibr" rid="ref64">64</xref>], 2026</td><td align="left" valign="top">Chest x-ray</td><td align="left" valign="top">Image to report</td><td align="left" valign="top">Human-supervised drafting</td><td align="left" valign="top">Workflow comparison</td><td align="left" valign="top">Report quality, efficiency</td></tr><tr><td align="left" valign="top">Tan [<xref ref-type="bibr" rid="ref65">65</xref>], 2026</td><td align="left" valign="top">Mixed modality</td><td align="left" valign="top">Report drafting or structuring</td><td align="left" valign="top">Human-supervised drafting</td><td align="left" valign="top">Implementation assessment</td><td align="left" valign="top">Acceptance, efficiency</td></tr><tr><td align="left" valign="top">Wang et al [<xref ref-type="bibr" rid="ref66">66</xref>], 2024</td><td align="left" valign="top">Ultrasound</td><td align="left" valign="top">Report drafting or structuring</td><td align="left" valign="top">Report editing or structuring support</td><td align="left" valign="top">Human expert assessment</td><td align="left" valign="top">Clinician feedback, accuracy, efficiency</td></tr><tr><td align="left" valign="top">Hong et al [<xref ref-type="bibr" rid="ref67">67</xref>], 2026</td><td align="left" valign="top">Chest x-ray</td><td align="left" valign="top">Image to report</td><td align="left" valign="top">Autonomous generation</td><td align="left" valign="top">Robustness and sensitivity analysis</td><td align="left" valign="top">Acceptance, safety, accuracy</td></tr><tr><td align="left" valign="top">Zanardo et al [<xref ref-type="bibr" rid="ref68">68</xref>], 2026</td><td align="left" valign="top">MRI</td><td align="left" valign="top">Report drafting or structuring</td><td align="left" valign="top">Autonomous generation</td><td align="left" valign="top">Human expert assessment</td><td align="left" valign="top">Report quality</td></tr><tr><td align="left" valign="top">Xie et al [<xref ref-type="bibr" rid="ref69">69</xref>], 2025</td><td align="left" valign="top">MRI</td><td align="left" valign="top">Report drafting or structuring</td><td align="left" valign="top">Report editing or structuring support</td><td align="left" valign="top">Human expert assessment</td><td align="left" valign="top">Acceptance, accuracy, efficiency</td></tr><tr><td align="left" valign="top">Wo&#x017A;nicki et al [<xref ref-type="bibr" rid="ref70">70</xref>], 2025</td><td align="left" valign="top">Chest x-ray</td><td align="left" valign="top">Report drafting or structuring</td><td align="left" valign="top">Report editing or structuring support</td><td align="left" valign="top">Reference standard comparison</td><td align="left" valign="top">Report quality, accuracy</td></tr><tr><td align="left" valign="top">Shao et al [<xref ref-type="bibr" rid="ref71">71</xref>], 2025</td><td align="left" valign="top">Ophthalmic imaging</td><td align="left" valign="top">Image to report</td><td align="left" valign="top">Autonomous generation</td><td align="left" valign="top">Human expert assessment</td><td align="left" valign="top">Report quality</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Categories are standardized for scanability and do not imply comparable outcome definitions across studies. Human-supervised drafting denotes AI involvement during report drafting, whereas report editing or structuring support denotes AI use after source report text, findings, or draft content were already available.</p></fn><fn id="table2fn2"><p><sup>b</sup>MRI: magnetic resonance imaging.</p></fn><fn id="table2fn3"><p><sup>c</sup>CT: computed tomography.</p></fn><fn id="table2fn4"><p><sup>d</sup>CTA: computed tomography angiography.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Risk of Bias and Certainty of Evidence</title><p>The risk-of-bias assessment framework is provided in Supplementary Table S3A in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. Overall risk of bias was moderate in 15 studies, high in 72, and serious in 14; no study was at low risk (<xref ref-type="table" rid="table3">Table 3</xref>). Although the evidence base included multicenter, external validation, and prospective validation studies, most studies were vulnerable to retrospective selection, incomplete external validation, nonrandomized workflow or reader comparisons, incomplete failure reporting, limited transparency in prompting or model handling, and selective emphasis on benchmark outcomes. Recurrent custom-framework concerns involved validation independence or leakage risk, absent or unclear blinding, heterogeneous outcome definitions, and poor accounting for failed or unusable reports. Risk-of-bias judgments are summarized in <xref ref-type="table" rid="table3">Table 3</xref>; complete study-level judgments are provided in Supplementary Table S3B in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>; and domain-level bias patterns are shown in <xref ref-type="fig" rid="figure3">Figure 3C</xref>.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Risk-of-bias summary by assessment domain<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Assessment set and domain</td><td align="left" valign="bottom">Risk-of-bias judgment distribution, n/N (%)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">All studies</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Overall</td><td align="left" valign="top">Low: 0; moderate: 15/101 (14.9); high: 72/101 (71.3); serious: 14/101 (13.9)</td></tr><tr><td align="left" valign="top" colspan="2">Custom AI validation+CLAIM<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Overall</td><td align="left" valign="top">Low: 0; moderate: 12/68 (17.6); high: 54/68 (79.4); serious: 2/68 (2.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Case selection</td><td align="left" valign="top">Low: 0; moderate: 11/68 (16.2); high: 55/68 (80.9); serious: 2/68 (2.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Reference standard</td><td align="left" valign="top">Low: 0; moderate: 46/68 (67.6); high: 22/68 (32.4); serious: 0</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Blinding</td><td align="left" valign="top">Low: 1/68 (1.5); moderate: 6/68 (8.8); high: 61/68 (89.7); serious: 0</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Validation or leakage</td><td align="left" valign="top">Low: 0; moderate: 9/68 (13.2); high: 57/68 (83.8); serious: 2/68 (2.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Model or prompt transparency</td><td align="left" valign="top">Low: 0; moderate: 41/68 (60.3); high: 27/68 (39.7); serious: 0</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Failure handling</td><td align="left" valign="top">Low: 0; moderate: 0; high: 66/68 (97.1); serious: 2/68 (2.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Outcome reporting</td><td align="left" valign="top">Low: 0; moderate: 18/68 (26.5); high: 50/68 (73.5); serious: 0</td></tr><tr><td align="left" valign="top" colspan="2">ROBINS-I<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup>+CLAIM</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Overall</td><td align="left" valign="top">Low: 0; moderate: 3/33 (9.1); high: 18/33 (54.5); serious: 12/33 (36.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Confounding</td><td align="left" valign="top">Low: 0; moderate: 3/33 (9.1); high: 18/33 (54.5); serious: 12/33 (36.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Participant selection</td><td align="left" valign="top">Low: 0; moderate: 23/33 (69.7); high: 0; serious: 10/33 (30.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Intervention classification</td><td align="left" valign="top">Low: 0; moderate: 33/33 (100); high: 0; serious: 0</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Intervention deviations</td><td align="left" valign="top">Low: 33/33 (100); moderate: 0; high: 0; serious: 0</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Missing data</td><td align="left" valign="top">Low: 0; moderate: 33/33 (100); high: 0; serious: 0</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Outcome measurement</td><td align="left" valign="top">Low: 0; moderate: 25/33 (75.8); high: 0; serious: 8/33 (24.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Reported result selection</td><td align="left" valign="top">Low: 0; moderate: 33/33 (100); high: 0; serious: 0</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>This table summarizes risk-of-bias judgments across the 101-study evidence base. Complete study-level judgments are provided in Supplementary Table S3B in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></fn><fn id="table3fn2"><p><sup>b</sup>CLAIM: Checklist for Artificial Intelligence in Medical Imaging.</p></fn><fn id="table3fn3"><p><sup>c</sup>ROBINS-I: Risk of Bias in Non-Randomized Studies of Interventions.</p></fn></table-wrap-foot></table-wrap><p>Certainty was judged very low across all 4 clinically important outcome domains in the GRADE Summary of Findings table (<xref ref-type="table" rid="table4">Table 4</xref>). This rating does not mean that all findings were clinically unfavorable; rather, it reflects a high or serious risk of bias, incompatible outcome definitions, indirectness across task types and clinical sources, sparse or nonreconstructable effect data, and concern that clinically unfavorable or failed report-generation outputs may be incompletely reported.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>GRADE<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup> summary of findings for clinically important outcome domains<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup>.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Outcome</td><td align="left" valign="bottom">Studies (n)</td><td align="left" valign="bottom">Study design</td><td align="left" valign="bottom">Risk of bias</td><td align="left" valign="bottom">Inconsistency</td><td align="left" valign="bottom">Indirectness</td><td align="left" valign="bottom">Imprecision</td><td align="left" valign="bottom">Other considerations</td><td align="left" valign="bottom">Impact</td><td align="left" valign="bottom">Certainty of the evidence (GRADE)</td><td align="left" valign="bottom">Importance</td></tr></thead><tbody><tr><td align="left" valign="top">Expert acceptance</td><td align="left" valign="top">6</td><td align="left" valign="top">Nonrandomized studies</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Strongly suspected publication or reporting bias</td><td align="left" valign="top">Not pooled. Acceptance measures varied; one clustered reader study reported acceptance of 70.5% (6047/8580) for AI-generated reports compared with 73.3% (6288/8580) for radiologist reports.</td><td align="left" valign="top">Very low</td><td align="left" valign="top">Important</td></tr><tr><td align="left" valign="top">Blinded expert preference</td><td align="left" valign="top">2</td><td align="left" valign="top">Nonrandomized studies</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Strongly suspected publication or reporting bias</td><td align="left" valign="top">Not pooled. One study reported AI-generated reports as equivalent or preferred in 77.7% (233/300) and 56.1% (170/303) of 2 datasets; another study reported that 74.2% (49/66) were indistinguishable from human-written reports.</td><td align="left" valign="top">Very low</td><td align="left" valign="top">Important</td></tr><tr><td align="left" valign="top">Safety outcomes (clinically significant errors, omission errors, and commission errors)</td><td align="left" valign="top">6</td><td align="left" valign="top">Nonrandomized studies</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Strongly suspected publication or reporting bias</td><td align="left" valign="top">Not pooled. Error definitions and denominators varied; one clustered reader study reported false-negative ratings of 18.5% (1584/8580) with AI compared with 17.8% (1527/8580) for radiologist reports, and false-positive ratings of 11.3% (971/8580) compared with 9.7% (831/8580).</td><td align="left" valign="top">Very low</td><td align="left" valign="top">Critical</td></tr><tr><td align="left" valign="top">Workflow burden</td><td align="left" valign="top">11</td><td align="left" valign="top">Nonrandomized studies</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Strongly suspected publication or reporting bias</td><td align="left" valign="top">Not pooled. Workflow effects were mixed: AI shortened MRI<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup> reading time (53 vs 61 s) but increased impression-editing time (18.29 vs 12.20 s) and edit distance (12.32 vs 5.74 words) in another study.</td><td align="left" valign="top">Very low</td><td align="left" valign="top">Important</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>GRADE: Grading of Recommendations Assessment, Development, and Evaluation.</p></fn><fn id="table4fn2"><p><sup>b</sup>This table follows the GRADEpro (Evidence Prime Inc) or GRADE Summary of Findings structure for narrative evidence. The full review included 101 studies; 36 reported at least one clinically relevant human evaluation, safety, workflow, report quality, acceptability, or time signal; the GRADEpro table uses the 21 selected studies with clinically interpretable end points, as shown in <xref ref-type="table" rid="table2">Table 2</xref>.</p></fn><fn id="table4fn3"><p><sup>c</sup>MRI: magnetic resonance imaging.&#xFEFF;</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-4"><title>Clinically Relevant Safety and Workflow Outcomes</title><p><xref ref-type="table" rid="table2">Table 2</xref> summarizes studies with clinically relevant human evaluation, safety, or workflow findings using standardized outcome domains, while key extractable quantitative findings are reported in this section, where they can be interpreted with study-specific context. In a large chest radiograph study of autonomous AI preliminary reports, acceptance was 70.5% (6047/8580) for AI-generated reports versus 73.3% (6288/8580) for radiologist reports, whereas false-positive findings were 11.3% (1527/8580) versus 9.7% (831/8580) and false-negative findings were 18.5% (1584/8580) versus 17.8% (1527/8580) in clustered reader evaluations [<xref ref-type="bibr" rid="ref51">51</xref>]. These evaluations were repeated reader-report or reader-case ratings rather than independent case-level events. In a clinician collaboration study for chest radiograph report generation, AI-generated reports were equivalent to or preferred over clinician reports in 77.7% (233/300) of one dataset and 56.1% (170/303) of another when judged by at least half of the raters, but clinically significant errors occurred in both AI-generated and clinician-generated reports [<xref ref-type="bibr" rid="ref52">52</xref>]. Outcome-level extraction details are provided in Supplementary Table S4 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p><p>Other human evaluation studies of autonomous multimodal generation reported expert indistinguishability, acceptance, or quality ratings in 3D brain CT report generation, bronchoscopy report generation, and medical vision&#x2013;language reporting across mixed devices [<xref ref-type="bibr" rid="ref55">55</xref>-<xref ref-type="bibr" rid="ref57">57</xref>]. Related autonomous report generation studies in chest x-ray, MRI, and fine-tuned LLaMA-based pipelines reported technical performance in additional settings [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref72">72</xref>].</p><p>Workflow outcomes differed across task types. In a brain MRI reader study, AI assistance improved diagnostic performance and reduced reading time from 61 to 53 seconds, with greater reported benefit among junior radiologists [<xref ref-type="bibr" rid="ref53">53</xref>]. In contrast, in a study of radiological impression drafting with multiple readers, drafts generated by the model required more editing time than the radiologist baseline (18.29 vs 12.20 s) and a greater edit distance (12.32 vs 5.74 words changed) [<xref ref-type="bibr" rid="ref54">54</xref>]. For this study of impression drafting, arm-specific means and CIs were sufficient to support the cautious quantitative reconstruction of single studies for sensitivity purposes, but no second clinically comparable workflow study was available for pooling (Supplementary Table S5 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>) [<xref ref-type="bibr" rid="ref54">54</xref>]. Additional workflow studies in lumbar spine MRI, breast ultrasound, structured positron emission tomography/CT reporting, and practical drafting with ChatGPT reported on workflow integration in heterogeneous settings [<xref ref-type="bibr" rid="ref73">73</xref>-<xref ref-type="bibr" rid="ref76">76</xref>].</p><p>Additional studies in the evidence base evaluated blinded radiologist assessment, structured report reader experiments, emergency CT impression drafting with assessment of critical omissions, real-world or pilot reporting workflows, prospective endoscopy report validation, and reporting time comparisons with LLM assistance (<xref ref-type="table" rid="table2">Table 2</xref>). Reported findings varied by task: some studies reported improved reporting efficiency, structured report usability, or expert acceptability, whereas others reported persistent omissions, hallucinations, lower quality ratings from specialists compared with human reports, or workflow benefits limited to narrow, standardized settings. The characteristics of studies added to the final evidence set are summarized in Supplementary Table S1 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref> [<xref ref-type="bibr" rid="ref58">58</xref>-<xref ref-type="bibr" rid="ref71">71</xref>,<xref ref-type="bibr" rid="ref77">77</xref>-<xref ref-type="bibr" rid="ref130">130</xref>].</p></sec><sec id="s3-5"><title>Feasibility of Meta-Analysis and Narrative Synthesis</title><p>No prespecified effectiveness, safety, or workflow burden outcome had at least 2 studies with directly comparable definitions, coherent designs, and analyzable effect structures suitable for meta-analysis. The expanded evidence base broadened clinical contexts but also added heterogeneity in inputs, outputs, comparators, human oversight, and denominator definitions. The strongest quantitative workflow signal came from the single impression-drafting study in <xref ref-type="table" rid="table2">Table 2</xref> and Supplementary Table S5 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. Safety and expert evaluation outcomes were usually reported as clustered reader-level counts, threshold preferences, paired overlap summaries, qualitative ratings, or pilot metrics that could not be pooled. These reporting barriers were summarized in Supplementary Table S6 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref> and structured the narrative synthesis.</p><p>No additional unpublished outcome data were available by May 19, 2026; therefore, all quantitative reconstruction and narrative synthesis relied on publicly reported results.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This systematic review assessed the effectiveness (expert acceptance and blinded preference), safety (clinically significant errors, omission errors, and commission errors), and workflow burden (reporting time, corrections, edit distance, and editing burden) of LLM-based medical report generation. Relative to these objectives, the principal finding is that the field is active and clinically diverse, but the reported evidence remains too heterogeneous to support stable conclusions about overall effectiveness, safety, or workflow benefit. Effectiveness outcomes suggested possible assistive value in selected settings [<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref52">52</xref>], but these outcomes were not consistently paired with case-level safety end points. Safety and workflow burden outcomes were reported with incompatible denominators, task structures, and comparison designs [<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref54">54</xref>].</p><p>The findings, therefore, support a cautious, task-specific interpretation rather than a general claim that LLM-based report generation is effective or safe. Some supervised workflows appeared acceptable to experts or were useful for selected reporting tasks, but current evidence does not justify autonomous replacement of clinicians [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref131">131</xref>]. Benefit and risk differed by task and workflow because impression generation from findings, human-supervised drafting, report editing or structuring, and autonomous image-to-report generation involve different clinical inputs, opportunities for human correction, and failure modes. These task categories should not be treated as interchangeable when interpreting acceptance, reporting errors, workflow burden, or implementation readiness.</p></sec><sec id="s4-2"><title>Interpretation and Comparison With Existing Literature</title><p>Prior reviews and commentaries have described the rapid emergence of AI report generation, structured reporting, prompt engineering, and foundation-model applications in imaging and endoscopy [<xref ref-type="bibr" rid="ref24">24</xref>-<xref ref-type="bibr" rid="ref28">28</xref>]. Recent systematic and scoping reviews have further summarized automated radiology report generation, structured reporting approaches, and deep learning trends [<xref ref-type="bibr" rid="ref29">29</xref>-<xref ref-type="bibr" rid="ref31">31</xref>], as well as LLM use in radiology reports, readability, simplification, and broader radiology adoption trajectories [<xref ref-type="bibr" rid="ref32">32</xref>-<xref ref-type="bibr" rid="ref34">34</xref>]. These publications establish that the technical literature is active, but leave a practical clinical question unresolved: whether published studies report clinically interpretable effectiveness, safety, and workflow outcomes in a way that can guide supervised implementation. This review differs from those prior reviews by prioritizing outcome definitions, the level of human oversight, and synthesis feasibility rather than technical capability alone. This emphasis allowed us to map not only what models can generate but also how reported outcomes relate to clinical usefulness.</p><p>The effectiveness findings need to be interpreted according to the outcomes that were actually reported. Expert acceptance, blinded preference, and report-quality judgments were the most common effectiveness signals, and several studies suggested that generated reports or AI-assisted reports could be acceptable, equivalent, or preferred in selected reader-evaluation settings [<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref52">52</xref>]. However, acceptance and preference are not the same as clinical effectiveness unless they are accompanied by evidence that final reports are accurate, clinically complete, and not more burdensome to verify [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. In this review, positive effectiveness signals were often limited by single-study designs, clustered reader ratings, heterogeneous rating scales, and incomplete linkage to omissions, commissions, or final clinician-edited report quality. Therefore, the evidence supports a narrower conclusion: LLM-generated or LLM-assisted report text may be useful in some supervised contexts, but published outcomes do not yet establish a generalizable effectiveness advantage across medical reporting workflows.</p><p>The safety and workflow outcomes further explain why broad effectiveness claims would be premature. In chest x-ray and clinician-collaboration settings, expert acceptance or preference could coexist with false-negative findings or clinically significant errors, showing that acceptable language does not necessarily imply safe reporting [<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref52">52</xref>]. Workflow findings were also mixed: a brain MRI study reported shorter reading time with AI assistance, whereas an impression-drafting study reported longer editing time and greater edit distance than the radiologist baseline [<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref54">54</xref>]. These findings suggest that workflow benefit depends on where the system enters the reporting process, whether clinicians start from images, findings, or draft text, and how much verification or correction the AI output requires.</p><p>The nonpoolability of the evidence base is itself an important synthesis finding. Meta-analysis was inappropriate not because publication volume was small, but because the evidence structure was incompatible across studies. Clustered reader-level counts, threshold preference summaries, paired workflow results without reconstructible dispersion, heterogeneous quality scales, pilot outcomes, and unclear denominators prevented valid pooling. This pattern aligns with SWiM principles: when effect estimates cannot be validly combined, transparent, structured narrative synthesis is preferable to forced quantitative aggregation [<xref ref-type="bibr" rid="ref42">42</xref>]. The expanded search increased clinical breadth but did not create a coherent synthesis set for any primary or key secondary outcome, indicating a reporting-standardization problem that limits cumulative clinical inference.</p><p>The dominance of chest x-ray studies also needs careful interpretation. Public corpora such as MIMIC-CXR have accelerated model development and benchmarking [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>], but repeated use of the same or related datasets can increase the risk of overlapping test sets, model and data leakage, and overly optimistic generalizability. This concern is especially relevant when training, validation, and external testing boundaries are not reported clearly. Endoscopy, pathology whole-slide imaging, CT, MRI, and ultrasound reporting involve different information densities, reference standard structures, and workflow constraints; consequently, evidence from templated chest radiography should not be generalized uncritically to more complex reporting domains.</p><p>The safety findings also require interpretation at the level of the clinical task. Human preference or acceptance of generated text should not be interpreted as equivalent to safety unless omissions, commissions, clinically significant discrepancies, and failed generations are measured explicitly [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. A fluent report can still omit an important abnormality, introduce an unsupported finding, or increase the review burden for the clinician who must verify it [<xref ref-type="bibr" rid="ref21">21</xref>-<xref ref-type="bibr" rid="ref23">23</xref>]. This distinction is especially important for autonomous image-to-report systems, where errors may arise before a clinician&#x2019;s review, but it also matters for impression-drafting and report-structuring tools because they can shape the language and emphasis of the final report. Accordingly, implementation decisions should require direct measurement of error types in the specific workflow under consideration.</p></sec><sec id="s4-3"><title>Clinical and Research Implications</title><p>For clinical practice, the real-world implication is that LLM-based reporting systems should be evaluated as supervised assistive tools, not as unsupervised replacements [<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref131">131</xref>]. Implementation decisions should be workflow specific because an AI tool that drafts an impression from existing findings has a different safety profile from an autonomous image-to-report model, and a tool that improves readability or structure does not necessarily reduce diagnostic error. Local validation should measure the task actually being deployed, the final clinician-edited report, and the workload created by reviewing or correcting generated text [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref131">131</xref>]. This means evaluating systems in the local modality, report template, patient mix, clinician review process, and information-technology environment in which they would actually be used.</p><p>For developers and clinical investigators, these findings point toward a staged evidence pathway. Benchmark metrics may be appropriate for early technical screening, but progression toward clinical use should require prospective or externally validated studies with safety outcomes at the case level and reconstructable workflow statistics [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref131">131</xref>]. Evaluation should separate omissions from commissions, report failed or unusable generations, state case-level denominators, and measure reporting time, corrections, edit distance, acceptance, blinded preference, and final report quality in realistic workflows [<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref131">131</xref>]. These reporting elements would make future evidence more useful for health systems because they connect apparent effectiveness to the safety and labor required to produce a clinically usable final report.</p><p>For evidence synthesis, the field needs more consistent reporting before meta-analysis can become informative. Future studies should define whether the unit of analysis is the patient, examination, report, reader-report pair, or generated text segment; specify whether evaluators were blinded; report how unusable generations were handled; and provide enough event counts or summary statistics to reconstruct effects [<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref131">131</xref>]. They should also report effectiveness, safety, and workflow outcomes together rather than presenting acceptance, errors, and time savings as isolated signals. Without these elements, additional studies may increase publication volume without improving certainty or allowing health systems to judge whether LLM-assisted reporting improves outcomes that matter in practice.</p></sec><sec id="s4-4"><title>Limitations</title><p>This review has limitations. The evidence base was dominated by retrospective studies; prospective workflow research was sparse; and no main outcome supported a robust pooled estimate across multiple directly comparable studies. The custom AI validation framework was prespecified and transparently domain-based, but it was not a formally validated measurement instrument, and interrater agreement statistics were not generated; accordingly, those judgments should be interpreted as structured qualitative bias signals rather than as validated scale outputs. The narrative GRADE-informed certainty assessment was likewise constrained by the absence of pooled estimates; it used GRADE domains to structure certainty judgments but did not produce certainty ratings around pooled effect sizes [<xref ref-type="bibr" rid="ref45">45</xref>].</p><p>Excluding preprints may underrepresent the newest model capabilities in a fast-moving field, but this choice was appropriate for a clinical review focused on peer-reviewed evidence. Some included and excluded studies may share public datasets or related benchmarks, reducing independence across studies [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>]. We also could not always determine whether model training data overlapped with evaluation data, especially for studies using public radiology corpora or proprietary foundation models. Finally, structured narrative synthesis provides less statistical compression than meta-analysis but better reflects the current nonpoolable evidence structure [<xref ref-type="bibr" rid="ref42">42</xref>].</p></sec><sec id="s4-5"><title>Conclusions and Broader Implications</title><p>This review provides a clinically oriented synthesis of LLM-based medical report generation across effectiveness, safety, workflow burden, and human oversight. It differs from radiology-specific, readability-oriented, and benchmark-focused reviews by asking whether peer-reviewed evidence can support clinical judgments about expert acceptance, blinded preference, clinically significant errors, omissions, commissions, reporting time, corrections, edit distance, and editing burden. Its main contribution is to show that the barrier is no longer simply whether models can produce plausible reports; it is whether studies report acceptance, preference, omissions, commissions, failed generations, time, corrections, edit distance, and human oversight well enough to guide practice. The broader real-world implication is that adoption should remain assistive, clinician-supervised, locally validated, and specific to the intended reporting workflow until stronger prospective safety and workflow evidence is available [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref131">131</xref>]. In practical terms, LLM-based reporting should be viewed as a candidate workflow aid whose value depends on the clinical task, validation setting, and human review process, not as a general replacement for expert report generation. The next stage of research should therefore move from benchmark-centered demonstrations toward standardized, case-level effectiveness, safety, and workflow burden evaluations that can support both local implementation decisions and future quantitative synthesis.</p></sec></sec></body><back><ack><p>The authors thank the corresponding authors of eligible studies who were contacted for clarification of potentially analyzable outcomes.</p><p>During the preparation of this manuscript, the authors used ChatGPT (GPT-5.5) to improve English language, grammar, and readability. The tool was not used for data analysis, interpretation of results, generation of scientific content, or reference retrieval. All AI-assisted output was reviewed, edited, and approved by the authors, who take full responsibility for the final content of the manuscript.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the National High-Level Hospital Clinical Research Funding (grant: LC2024A04) and the Chinese Academy of Medical Sciences Innovation Fund for Medical Sciences (grant: 2022-I2M-C&#x0026;T-B-059).</p></sec><sec><title>Data Availability</title><p>All data extracted and analyzed in this systematic review are presented in the article and its multimedia appendices. The prespecified extraction framework, study-level extraction sheet, verified outcome-level extraction sheet, and conversion decision log are available in the <xref ref-type="supplementary-material" rid="app1">Multimedia Appendices 1</xref> and <xref ref-type="supplementary-material" rid="app2">2</xref>; additional clarifying materials are available from the corresponding author upon reasonable request.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: JLH, XGN</p><p>Methodology: JLH, JQZ, XGN</p><p>Literature screening: JLH, JQZ</p><p>Data curation: JLH</p><p>Formal analysis: JLH, JQZ</p><p>Investigation: JLH, JQZ</p><p>Supervision: XGN</p><p>Writing &#x2013; original draft: JLH</p><p>Writing &#x2013; review &#x0026; editing: JQZ, XGN</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">BERT</term><def><p>Bidirectional Encoder Representations from Transformers</p></def></def-item><def-item><term id="abb2">BLEU</term><def><p>Bilingual Evaluation Understudy</p></def></def-item><def-item><term id="abb3">CIDEr</term><def><p>Consensus-Based Image Description Evaluation</p></def></def-item><def-item><term id="abb4">CLAIM</term><def><p>Checklist for Artificial Intelligence in Medical Imaging</p></def></def-item><def-item><term id="abb5">CT</term><def><p>computed tomography</p></def></def-item><def-item><term id="abb6">GRADE</term><def><p>Grading of Recommendations Assessment, Development, and Evaluation</p></def></def-item><def-item><term id="abb7">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb8">MRI</term><def><p>magnetic resonance imaging</p></def></def-item><def-item><term id="abb9">PACS</term><def><p>picture archiving and communication system</p></def></def-item><def-item><term id="abb10">PRISMA</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses</p></def></def-item><def-item><term id="abb11">PRISMA-S</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses Literature Search Extension</p></def></def-item><def-item><term id="abb12">PROSPERO</term><def><p> International Prospective Register of Systematic Reviews</p></def></def-item><def-item><term id="abb13">ROBINS-I</term><def><p>Risk of Bias in Non-Randomized Studies of Interventions</p></def></def-item><def-item><term id="abb14">ROUGE</term><def><p>Recall-Oriented Understudy for Gisting Evaluation</p></def></def-item><def-item><term id="abb15">SWiM</term><def><p>Synthesis Without Meta-Analysis</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sinsky</surname><given-names>C</given-names> </name><name name-style="western"><surname>Colligan</surname><given-names>L</given-names> </name><name name-style="western"><surname>Li</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Allocation of physician time in ambulatory practice: a time and motion study in 4 specialties</article-title><source>Ann Intern Med</source><year>2016</year><month>12</month><day>6</day><volume>165</volume><issue>11</issue><fpage>753</fpage><lpage>760</lpage><pub-id pub-id-type="doi">10.7326/M16-0961</pub-id><pub-id pub-id-type="medline">27595430</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jing</surname><given-names>B</given-names> </name><name name-style="western"><surname>Xie</surname><given-names>P</given-names> </name><name name-style="western"><surname>Xing</surname><given-names>E</given-names> </name></person-group><article-title>On the automatic generation of medical imaging reports</article-title><source>Proc Assoc Comput Linguist Annu Meet</source><year>2018</year><fpage>2577</fpage><lpage>2586</lpage><pub-id pub-id-type="doi">10.18653/v1/P18-1240</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Song</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>TH</given-names> </name><name name-style="western"><surname>Wan</surname><given-names>X</given-names> </name></person-group><article-title>Generating radiology reports via memory-driven transformer</article-title><source>Proc Conf Empir Methods Nat Lang Process</source><year>2020</year><fpage>1439</fpage><lpage>1449</lpage><pub-id pub-id-type="doi">10.18653/v1/2020.emnlp-main.112</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhong</surname><given-names>T</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>W</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>ChatRadio-Valuer: a chat large language model for generalizable radiology impression generation on multi-institution and multi-system data</article-title><source>IEEE Trans Biomed Eng</source><year>2026</year><month>03</month><volume>73</volume><issue>3</issue><fpage>1050</fpage><lpage>1061</lpage><pub-id pub-id-type="doi">10.1109/TBME.2025.3597325</pub-id><pub-id pub-id-type="medline">40788800</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yeasin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Moinuddin</surname><given-names>KA</given-names> </name><name name-style="western"><surname>Havugimana</surname><given-names>F</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Park</surname><given-names>P</given-names> </name></person-group><article-title>Auto-Rad: end-to-end report generation from lumber spine MRI using vision-language model</article-title><source>J Clin Med</source><year>2024</year><month>11</month><day>23</day><volume>13</volume><issue>23</issue><fpage>7092</fpage><pub-id pub-id-type="doi">10.3390/jcm13237092</pub-id><pub-id pub-id-type="medline">39685549</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>RW</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>KH</given-names> </name><name name-style="western"><surname>Yun</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Choi</surname><given-names>HS</given-names> </name></person-group><article-title>Comparative analysis of M4CXR, an LLM-based chest X-ray report generation model, and ChatGPT in radiological interpretation</article-title><source>J Clin Med</source><year>2024</year><month>11</month><day>22</day><volume>13</volume><issue>23</issue><fpage>7057</fpage><pub-id pub-id-type="doi">10.3390/jcm13237057</pub-id><pub-id pub-id-type="medline">39685515</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Bie</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>H</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>H</given-names> </name></person-group><article-title>Large language model with region-guided referring and grounding for CT report generation</article-title><source>IEEE Trans Med Imaging</source><year>2025</year><month>08</month><volume>44</volume><issue>8</issue><fpage>3139</fpage><lpage>3150</lpage><pub-id pub-id-type="doi">10.1109/TMI.2025.3559923</pub-id><pub-id pub-id-type="medline">40215158</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Massimi</surname><given-names>D</given-names> </name><name name-style="western"><surname>Stefano</surname><given-names>LD</given-names> </name><name name-style="western"><surname>Rizkala</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Large language model-driven analysis and report generation of endoscopy videos-A pilot study</article-title><source>Dig Endosc</source><year>2026</year><month>03</month><volume>38</volume><issue>3</issue><fpage>e70134</fpage><pub-id pub-id-type="doi">10.1111/den.70134</pub-id><pub-id pub-id-type="medline">41804240</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zheng</surname><given-names>X</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>A fine-tuning multimodal large language model for endoscopic report generation</article-title><source>Biomed Signal Process Control</source><year>2026</year><month>06</month><volume>118</volume><fpage>109737</fpage><pub-id pub-id-type="doi">10.1016/j.bspc.2026.109737</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Joyson</surname><given-names>JL</given-names> </name><name name-style="western"><surname>Soundar</surname><given-names>KR</given-names> </name><name name-style="western"><surname>Nancy</surname><given-names>P</given-names> </name></person-group><article-title>A multi-modal fusion-based deep learning with finetuned LLaMA 3 for lung disease diagnosis using PACS radiology reports</article-title><source>Biomed Signal Process Control</source><year>2026</year><month>05</month><volume>117</volume><fpage>109466</fpage><pub-id pub-id-type="doi">10.1016/j.bspc.2026.109466</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>K</given-names> </name><etal/></person-group><article-title>MetaGP: a generative foundation model integrating electronic health records and multimodal imaging for addressing unmet clinical needs</article-title><source>Cell Rep Med</source><year>2025</year><month>04</month><day>15</day><volume>6</volume><issue>4</issue><fpage>102056</fpage><pub-id pub-id-type="doi">10.1016/j.xcrm.2025.102056</pub-id><pub-id pub-id-type="medline">40187356</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dasanayaka</surname><given-names>C</given-names> </name><name name-style="western"><surname>Dandeniya</surname><given-names>K</given-names> </name><name name-style="western"><surname>Dissanayake</surname><given-names>MB</given-names> </name><name name-style="western"><surname>Gunasena</surname><given-names>C</given-names> </name><name name-style="western"><surname>Jayasinghe</surname><given-names>R</given-names> </name></person-group><article-title>Multimodal AI and large language models for orthopantomography radiology report generation and Q&#x0026;A</article-title><source>Appl Syst Innov</source><year>2025</year><volume>8</volume><issue>2</issue><fpage>39</fpage><pub-id pub-id-type="doi">10.3390/asi8020039</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Helmy</surname><given-names>H</given-names> </name><name name-style="western"><surname>Hosseini</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ibrahim</surname><given-names>A</given-names> </name><name name-style="western"><surname>Baig-Mirza</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sadek</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Serag</surname><given-names>A</given-names> </name></person-group><article-title>SPINE: segmentation-guided processing and integration of multimodal spinal MRI for natural-language enhanced report generation</article-title><source>Appl Artif Intell</source><year>2026</year><volume>40</volume><issue>1</issue><pub-id pub-id-type="doi">10.1080/08839514.2026.2626117</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Li</surname><given-names>M</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>Q</given-names> </name></person-group><article-title>Ultrasound report generation with fuzzy knowledge and multi-modal large language model</article-title><source>Expert Syst Appl</source><year>2025</year><month>11</month><volume>292</volume><fpage>128555</fpage><pub-id pub-id-type="doi">10.1016/j.eswa.2025.128555</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>L</given-names> </name></person-group><article-title>R2GenGPT: radiology report generation with frozen LLMs</article-title><source>Meta-Radiology</source><year>2023</year><month>11</month><volume>1</volume><issue>3</issue><fpage>100033</fpage><pub-id pub-id-type="doi">10.1016/j.metrad.2023.100033</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>A systematic evaluation of GPT-4V&#x2019;s multimodal capability for chest X-ray image analysis</article-title><source>Meta-Radiology</source><year>2024</year><month>12</month><volume>2</volume><issue>4</issue><fpage>100099</fpage><pub-id pub-id-type="doi">10.1016/j.metrad.2024.100099</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>KA</given-names> </name><name name-style="western"><surname>Hong</surname><given-names>S</given-names> </name><name name-style="western"><surname>Yoo</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Shim</surname><given-names>HS</given-names> </name></person-group><article-title>Enhancing structured pathology report generation with foundation model and modular design</article-title><source>IEEE Access</source><year>2025</year><volume>13</volume><fpage>121290</fpage><lpage>121299</lpage><pub-id pub-id-type="doi">10.1109/ACCESS.2025.3588121</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yan</surname><given-names>H</given-names> </name><name name-style="western"><surname>Shao</surname><given-names>D</given-names> </name></person-group><article-title>Multimodal medical image analysis: integrating LLM and RAG deep learning strategies</article-title><source>J Adv Inf Technol</source><year>2025</year><volume>16</volume><issue>4</issue><fpage>568</fpage><lpage>581</lpage><pub-id pub-id-type="doi">10.12720/jait.16.4.568-581</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lyell</surname><given-names>D</given-names> </name><name name-style="western"><surname>Coiera</surname><given-names>E</given-names> </name></person-group><article-title>Automation bias and verification complexity: a systematic review</article-title><source>J Am Med Inform Assoc</source><year>2017</year><month>03</month><day>1</day><volume>24</volume><issue>2</issue><fpage>423</fpage><lpage>431</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocw105</pub-id><pub-id pub-id-type="medline">27516495</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Challen</surname><given-names>R</given-names> </name><name name-style="western"><surname>Denny</surname><given-names>J</given-names> </name><name name-style="western"><surname>Pitt</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gompels</surname><given-names>L</given-names> </name><name name-style="western"><surname>Edwards</surname><given-names>T</given-names> </name><name name-style="western"><surname>Tsaneva-Atanasova</surname><given-names>K</given-names> </name></person-group><article-title>Artificial intelligence, bias and clinical safety</article-title><source>BMJ Qual Saf</source><year>2019</year><month>03</month><volume>28</volume><issue>3</issue><fpage>231</fpage><lpage>237</lpage><pub-id pub-id-type="doi">10.1136/bmjqs-2018-008370</pub-id><pub-id pub-id-type="medline">30636200</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>P</given-names> </name><name name-style="western"><surname>Bubeck</surname><given-names>S</given-names> </name><name name-style="western"><surname>Petro</surname><given-names>J</given-names> </name></person-group><article-title>Benefits, limits, and risks of GPT-4 as an AI chatbot for medicine</article-title><source>N Engl J Med</source><year>2023</year><month>03</month><day>30</day><volume>388</volume><issue>13</issue><fpage>1233</fpage><lpage>1239</lpage><pub-id pub-id-type="doi">10.1056/NEJMsr2214184</pub-id><pub-id pub-id-type="medline">36988602</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ji</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>N</given-names> </name><name name-style="western"><surname>Frieske</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Survey of hallucination in natural language generation</article-title><source>ACM Comput Surv</source><year>2023</year><month>12</month><day>31</day><volume>55</volume><issue>12</issue><fpage>1</fpage><lpage>38</lpage><pub-id pub-id-type="doi">10.1145/3571730</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yi</surname><given-names>PH</given-names> </name><name name-style="western"><surname>Haver</surname><given-names>HL</given-names> </name><name name-style="western"><surname>Jeudy</surname><given-names>JJ</given-names> </name><etal/></person-group><article-title>Best practices for the safe use of large language models and other generative AI in radiology</article-title><source>Radiology</source><year>2025</year><month>09</month><volume>316</volume><issue>3</issue><fpage>e241516</fpage><pub-id pub-id-type="doi">10.1148/radiol.241516</pub-id><pub-id pub-id-type="medline">40985835</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Seah</surname><given-names>JCY</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>JSN</given-names> </name><name name-style="western"><surname>Tran</surname><given-names>A</given-names> </name></person-group><article-title>Drafting the future: the dawn of AI report generation in radiology</article-title><source>Radiology</source><year>2025</year><month>07</month><volume>316</volume><issue>1</issue><fpage>e243378</fpage><pub-id pub-id-type="doi">10.1148/radiol.243378</pub-id><pub-id pub-id-type="medline">40728403</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>TT</given-names> </name><name name-style="western"><surname>Makutonin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sirous</surname><given-names>R</given-names> </name><name name-style="western"><surname>Javan</surname><given-names>R</given-names> </name></person-group><article-title>Optimizing large language models in radiology and mitigating pitfalls: prompt engineering and fine-tuning</article-title><source>Radiographics</source><year>2025</year><month>04</month><volume>45</volume><issue>4</issue><fpage>e240073</fpage><pub-id pub-id-type="doi">10.1148/rg.240073</pub-id><pub-id pub-id-type="medline">40048389</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Oda</surname><given-names>M</given-names> </name></person-group><article-title>Generative AI and foundation models in medical image</article-title><source>Radiol Phys Technol</source><year>2025</year><month>12</month><volume>18</volume><issue>4</issue><fpage>937</fpage><lpage>948</lpage><pub-id pub-id-type="doi">10.1007/s12194-025-00968-1</pub-id><pub-id pub-id-type="medline">41051729</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Labaki</surname><given-names>C</given-names> </name><name name-style="western"><surname>Uche-Anya</surname><given-names>EN</given-names> </name><name name-style="western"><surname>Berzin</surname><given-names>TM</given-names> </name></person-group><article-title>Artificial intelligence in gastrointestinal endoscopy</article-title><source>Gastroenterol Clin North Am</source><year>2024</year><month>12</month><volume>53</volume><issue>4</issue><fpage>773</fpage><lpage>786</lpage><pub-id pub-id-type="doi">10.1016/j.gtc.2024.08.005</pub-id><pub-id pub-id-type="medline">39489586</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Biswas</surname><given-names>S</given-names> </name><name name-style="western"><surname>Khan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Awal</surname><given-names>SS</given-names> </name></person-group><article-title>Can ChatGPT write radiology reports?</article-title><source>Chin J Acad Radiol</source><year>2024</year><month>03</month><volume>7</volume><issue>1</issue><fpage>102</fpage><lpage>106</lpage><pub-id pub-id-type="doi">10.1007/s42058-023-00132-x</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sloan</surname><given-names>P</given-names> </name><name name-style="western"><surname>Clatworthy</surname><given-names>P</given-names> </name><name name-style="western"><surname>Simpson</surname><given-names>E</given-names> </name><name name-style="western"><surname>Mirmehdi</surname><given-names>M</given-names> </name></person-group><article-title>Automated radiology report generation: a review of recent advances</article-title><source>IEEE Rev Biomed Eng</source><year>2025</year><volume>18</volume><fpage>368</fpage><lpage>387</lpage><pub-id pub-id-type="doi">10.1109/RBME.2024.3408456</pub-id><pub-id pub-id-type="medline">38829752</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Busch</surname><given-names>F</given-names> </name><name name-style="western"><surname>Hoffmann</surname><given-names>L</given-names> </name><name name-style="western"><surname>Dos Santos</surname><given-names>DP</given-names> </name><etal/></person-group><article-title>Large language models for structured reporting in radiology: past, present, and future</article-title><source>Eur Radiol</source><year>2025</year><month>05</month><volume>35</volume><issue>5</issue><fpage>2589</fpage><lpage>2602</lpage><pub-id pub-id-type="doi">10.1007/s00330-024-11107-6</pub-id><pub-id pub-id-type="medline">39438330</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Izhar</surname><given-names>A</given-names> </name><name name-style="western"><surname>Idris</surname><given-names>N</given-names> </name><name name-style="western"><surname>Japar</surname><given-names>N</given-names> </name></person-group><article-title>Medical radiology report generation: a systematic review of current deep learning methods, trends, and future directions</article-title><source>Artif Intell Med</source><year>2025</year><month>10</month><volume>168</volume><fpage>103220</fpage><pub-id pub-id-type="doi">10.1016/j.artmed.2025.103220</pub-id><pub-id pub-id-type="medline">40700862</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>RC</given-names> </name><name name-style="western"><surname>Hadidchi</surname><given-names>R</given-names> </name><name name-style="western"><surname>Coard</surname><given-names>MC</given-names> </name><etal/></person-group><article-title>Use of large language models on radiology reports: a scoping review</article-title><source>J Am Coll Radiol</source><year>2026</year><month>03</month><volume>23</volume><issue>3</issue><fpage>437</fpage><lpage>454</lpage><pub-id pub-id-type="doi">10.1016/j.jacr.2025.10.005</pub-id><pub-id pub-id-type="medline">41196263</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Patwardhan</surname><given-names>V</given-names> </name><name name-style="western"><surname>Balchander</surname><given-names>D</given-names> </name><name name-style="western"><surname>Fussell</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Leveraging large language models to enhance radiology report readability: a systematic review</article-title><source>J Am Coll Radiol</source><year>2026</year><month>03</month><volume>23</volume><issue>3</issue><fpage>354</fpage><lpage>361</lpage><pub-id pub-id-type="doi">10.1016/j.jacr.2025.09.004</pub-id><pub-id pub-id-type="medline">40945554</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Al Zaabi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Alshibli</surname><given-names>R</given-names> </name><name name-style="western"><surname>AlAmri</surname><given-names>A</given-names> </name><name name-style="western"><surname>AlRuheili</surname><given-names>I</given-names> </name><name name-style="western"><surname>Lutfi</surname><given-names>SL</given-names> </name></person-group><article-title>Trends and trajectories in the rise of large language models in radiology: scoping review</article-title><source>JMIR Med Inform</source><year>2025</year><month>12</month><day>9</day><volume>13</volume><fpage>e78041</fpage><pub-id pub-id-type="doi">10.2196/78041</pub-id><pub-id pub-id-type="medline">41364806</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Papineni</surname><given-names>K</given-names> </name><name name-style="western"><surname>Roukos</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ward</surname><given-names>T</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>WJ</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Isabelle</surname><given-names>P</given-names> </name><name name-style="western"><surname>Charniak</surname><given-names>E</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>D</given-names> </name></person-group><article-title>BLEU: a method for automatic evaluation of machine translation</article-title><source>Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics</source><year>2002</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>311</fpage><lpage>318</lpage><pub-id pub-id-type="doi">10.3115/1073083.1073135</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>CY</given-names> </name></person-group><article-title>ROUGE: a package for automatic evaluation of summaries</article-title><source>Text Summarization Branches Out: Proceedings of the ACL-04 Workshop</source><year>2004</year><access-date>2026-08-01</access-date><publisher-name>Association for Computational Linguistics</publisher-name><fpage>74</fpage><lpage>81</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/W04-1013/">https://aclanthology.org/W04-1013/</ext-link></comment></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vedantam</surname><given-names>R</given-names> </name><name name-style="western"><surname>Zitnick</surname><given-names>CL</given-names> </name><name name-style="western"><surname>Parikh</surname><given-names>D</given-names> </name></person-group><article-title>CIDEr: consensus-based image description evaluation</article-title><source>Proc IEEE Comput Vis Pattern Recognit Conf (CVPR)</source><year>2015</year><fpage>4566</fpage><lpage>4575</lpage><pub-id pub-id-type="doi">10.1109/CVPR.2015.7299087</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>T</given-names> </name><name name-style="western"><surname>Kishore</surname><given-names>V</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Weinberger</surname><given-names>KQ</given-names> </name><name name-style="western"><surname>Artzi</surname><given-names>Y</given-names> </name></person-group><article-title>BERTScore: evaluating text generation with BERT</article-title><year>2020</year><access-date>2026-08-01</access-date><conf-name>International Conference on Learning Representations</conf-name><conf-date>Apr 26-30, 2020</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://openreview.net/pdf?id=SkeHuCVFDr">https://openreview.net/pdf?id=SkeHuCVFDr</ext-link></comment></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gholipour Picha</surname><given-names>S</given-names> </name><name name-style="western"><surname>Al Chanti</surname><given-names>D</given-names> </name><name name-style="western"><surname>Caplier</surname><given-names>A</given-names> </name></person-group><article-title>Trust but verify: image-aware evaluation of radiology report generators</article-title><source>Mach Learn Appl</source><year>2026</year><month>03</month><volume>23</volume><fpage>100851</fpage><pub-id pub-id-type="doi">10.1016/j.mlwa.2026.100851</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Page</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>McKenzie</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Bossuyt</surname><given-names>PM</given-names> </name><etal/></person-group><article-title>The PRISMA 2020 statement: an updated guideline for reporting systematic reviews</article-title><source>BMJ</source><year>2021</year><month>03</month><day>29</day><volume>372</volume><fpage>n71</fpage><pub-id pub-id-type="doi">10.1136/bmj.n71</pub-id><pub-id pub-id-type="medline">33782057</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rethlefsen</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Kirtley</surname><given-names>S</given-names> </name><name name-style="western"><surname>Waffenschmidt</surname><given-names>S</given-names> </name><etal/></person-group><article-title>PRISMA-S: an extension to the PRISMA statement for reporting literature searches in systematic reviews</article-title><source>Syst Rev</source><year>2021</year><month>01</month><day>26</day><volume>10</volume><issue>1</issue><fpage>39</fpage><pub-id pub-id-type="doi">10.1186/s13643-020-01542-z</pub-id><pub-id pub-id-type="medline">33499930</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Campbell</surname><given-names>M</given-names> </name><name name-style="western"><surname>McKenzie</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Sowden</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Synthesis without meta-analysis (SWiM) in systematic reviews: reporting guideline</article-title><source>BMJ</source><year>2020</year><month>01</month><day>16</day><volume>368</volume><fpage>l6890</fpage><pub-id pub-id-type="doi">10.1136/bmj.l6890</pub-id><pub-id pub-id-type="medline">31948937</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sterne</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Hern&#x00E1;n</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Reeves</surname><given-names>BC</given-names> </name><etal/></person-group><article-title>ROBINS-I: a tool for assessing risk of bias in non-randomised studies of interventions</article-title><source>BMJ</source><year>2016</year><month>10</month><day>12</day><volume>355</volume><fpage>i4919</fpage><pub-id pub-id-type="doi">10.1136/bmj.i4919</pub-id><pub-id pub-id-type="medline">27733354</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tejani</surname><given-names>AS</given-names> </name><name name-style="western"><surname>Klontzas</surname><given-names>ME</given-names> </name><name name-style="western"><surname>Gatti</surname><given-names>AA</given-names> </name><etal/></person-group><article-title>Checklist for artificial intelligence in medical imaging (CLAIM): 2024 update</article-title><source>Radiol Artif Intell</source><year>2024</year><month>07</month><volume>6</volume><issue>4</issue><fpage>e240300</fpage><pub-id pub-id-type="doi">10.1148/ryai.240300</pub-id><pub-id pub-id-type="medline">38809149</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guyatt</surname><given-names>GH</given-names> </name><name name-style="western"><surname>Oxman</surname><given-names>AD</given-names> </name><name name-style="western"><surname>Vist</surname><given-names>GE</given-names> </name><etal/></person-group><article-title>GRADE: an emerging consensus on rating quality of evidence and strength of recommendations</article-title><source>BMJ</source><year>2008</year><month>04</month><day>26</day><volume>336</volume><issue>7650</issue><fpage>924</fpage><lpage>926</lpage><pub-id pub-id-type="doi">10.1136/bmj.39489.470347.AD</pub-id><pub-id pub-id-type="medline">18436948</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>F</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Activating associative disease-aware vision token memory for LLM-based X-ray report generation</article-title><source>IEEE Trans Med Imaging</source><year>2026</year><month>02</month><volume>45</volume><issue>2</issue><fpage>583</fpage><lpage>595</lpage><pub-id pub-id-type="doi">10.1109/TMI.2025.3603416</pub-id><pub-id pub-id-type="medline">40864571</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Noh</surname><given-names>WJ</given-names> </name><name name-style="western"><surname>Pi</surname><given-names>SW</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>BD</given-names> </name></person-group><article-title>Hybrid framework for lesion-aware, clinically coherent chest X-ray report generation using contrastive learning and large language models</article-title><source>Sci Rep</source><year>2026</year><month>01</month><day>5</day><volume>16</volume><issue>1</issue><fpage>4645</fpage><pub-id pub-id-type="doi">10.1038/s41598-025-34799-2</pub-id><pub-id pub-id-type="medline">41492086</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Han</surname><given-names>P</given-names> </name><name name-style="western"><surname>Li</surname><given-names>X</given-names> </name><name name-style="western"><surname>Jing</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wei</surname><given-names>J</given-names> </name></person-group><article-title>Dynamic feature fusion guiding and multimodal large language model refining for medical image report generation</article-title><source>Expert Syst Appl</source><year>2026</year><month>03</month><volume>299</volume><fpage>130082</fpage><pub-id pub-id-type="doi">10.1016/j.eswa.2025.130082</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jia</surname><given-names>X</given-names> </name><name name-style="western"><surname>Xiong</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Pei</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>C</given-names> </name><name name-style="western"><surname>Fang</surname><given-names>Z</given-names> </name></person-group><article-title>Semantic feedback-based RAG for radiology report generation</article-title><source>Big Data Min Anal</source><year>2026</year><volume>9</volume><issue>2</issue><fpage>393</fpage><lpage>406</lpage><pub-id pub-id-type="doi">10.26599/BDMA.2025.9020037</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>F</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Li</surname><given-names>C</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>B</given-names> </name></person-group><article-title>R2GenCSR: mining contextual and residual information for LLM-based radiology report generation</article-title><source>IEEE J Biomed Health Inform</source><volume>2026</volume><fpage>1</fpage><lpage>14</lpage><pub-id pub-id-type="doi">10.1109/JBHI.2026.3669539</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hong</surname><given-names>EK</given-names> </name><name name-style="western"><surname>Ham</surname><given-names>J</given-names> </name><name name-style="western"><surname>Roh</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Diagnostic accuracy and clinical value of a domain-specific multimodal generative AI model for chest radiograph report generation</article-title><source>Radiology</source><year>2025</year><month>03</month><volume>314</volume><issue>3</issue><fpage>e241476</fpage><pub-id pub-id-type="doi">10.1148/radiol.241476</pub-id><pub-id pub-id-type="medline">40131111</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tanno</surname><given-names>R</given-names> </name><name name-style="western"><surname>Barrett</surname><given-names>DGT</given-names> </name><name name-style="western"><surname>Sellergren</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Collaboration between clinicians and vision-language models in radiology report generation</article-title><source>Nat Med</source><year>2025</year><month>02</month><volume>31</volume><issue>2</issue><fpage>599</fpage><lpage>608</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03302-1</pub-id><pub-id pub-id-type="medline">39511432</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>RP</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>WJ</given-names> </name><etal/></person-group><article-title>Evaluation of large language models for diagnostic impression generation from brain MRI report findings: a multicenter benchmark and reader study</article-title><source>NPJ Digit Med</source><year>2026</year><month>01</month><day>22</day><volume>9</volume><issue>1</issue><fpage>187</fpage><pub-id pub-id-type="doi">10.1038/s41746-026-02380-4</pub-id><pub-id pub-id-type="medline">41571872</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Serapio</surname><given-names>A</given-names> </name><name name-style="western"><surname>Chaudhari</surname><given-names>G</given-names> </name><name name-style="western"><surname>Savage</surname><given-names>C</given-names> </name><etal/></person-group><article-title>An open-source fine-tuned large language model for radiological impression generation: a multi-reader performance study</article-title><source>BMC Med Imaging</source><year>2024</year><month>09</month><day>27</day><volume>24</volume><issue>1</issue><fpage>254</fpage><pub-id pub-id-type="doi">10.1186/s12880-024-01435-w</pub-id><pub-id pub-id-type="medline">39333958</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>CY</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>KJ</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>CF</given-names> </name><etal/></person-group><article-title>Towards a holistic framework for multimodal LLM in 3D brain CT radiology report generation</article-title><source>Nat Commun</source><year>2025</year><month>03</month><day>6</day><volume>16</volume><issue>1</issue><fpage>2258</fpage><pub-id pub-id-type="doi">10.1038/s41467-025-57426-0</pub-id><pub-id pub-id-type="medline">40050277</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Luo</surname><given-names>X</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Towards automated reporting: a bronchoscopy report dataset for enhancing multimodality large language models</article-title><source>Sci Data</source><year>2026</year><month>02</month><day>3</day><volume>13</volume><issue>1</issue><fpage>339</fpage><pub-id pub-id-type="doi">10.1038/s41597-026-06692-8</pub-id><pub-id pub-id-type="medline">41634075</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ayaz</surname><given-names>M</given-names> </name><name name-style="western"><surname>Khan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Saqib</surname><given-names>M</given-names> </name><name name-style="western"><surname>Khelifi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sajjad</surname><given-names>M</given-names> </name><name name-style="western"><surname>Elsaddik</surname><given-names>A</given-names> </name></person-group><article-title>MedVLM: medical vision&#x2013;language model for consumer devices</article-title><source>IEEE Consum Electron Mag</source><year>2025</year><volume>14</volume><issue>5</issue><fpage>75</fpage><lpage>83</lpage><pub-id pub-id-type="doi">10.1109/MCE.2024.3522521</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>M</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Miao</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Fine-tuned large language model for automated radiology impression generation: a multicenter evaluation</article-title><source>Radiol Artif Intell</source><year>2026</year><month>05</month><volume>8</volume><issue>3</issue><fpage>e250714</fpage><pub-id pub-id-type="doi">10.1148/ryai.250714</pub-id><pub-id pub-id-type="medline">41983921</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>de Margerie-Mellon</surname><given-names>C</given-names> </name><name name-style="western"><surname>Duron</surname><given-names>L</given-names> </name><name name-style="western"><surname>Fournier</surname><given-names>L</given-names> </name><name name-style="western"><surname>Ouakil</surname><given-names>L</given-names> </name><name name-style="western"><surname>Soulat</surname><given-names>G</given-names> </name><name name-style="western"><surname>Teixeira</surname><given-names>PG</given-names> </name></person-group><article-title>Reporting efficiency in diagnostic imaging: can plug-and-play general-purpose large language models outperform conventional speech recognition?</article-title><source>Eur Radiol</source><year>2026</year><month>08</month><volume>36</volume><issue>8</issue><fpage>6282</fpage><lpage>6291</lpage><pub-id pub-id-type="doi">10.1007/s00330-026-12524-5</pub-id><pub-id pub-id-type="medline">42009869</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jiang</surname><given-names>R</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>B</given-names> </name><name name-style="western"><surname>Dong</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Domain specific multimodal large language model for automated endoscopy reporting with multicenter prospective validation</article-title><source>NPJ Digit Med</source><year>2026</year><volume>9</volume><issue>1</issue><fpage>394</fpage><pub-id pub-id-type="doi">10.1038/s41746-026-02569-7</pub-id><pub-id pub-id-type="medline">41904204</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tekdemir</surname><given-names>H</given-names> </name><name name-style="western"><surname>&#x00C7;&#x0131;vg&#x0131;n</surname><given-names>E</given-names> </name><name name-style="western"><surname>Akp&#x0131;nar</surname><given-names>&#x015E;</given-names> </name><etal/></person-group><article-title>Evaluating large language models for Turkish emergency CT impression drafting: quality, critical omissions, and readability</article-title><source>J Imaging Inform Med</source><year>2026</year><month>05</month><day>6</day><pub-id pub-id-type="doi">10.1007/s10278-026-01989-x</pub-id><pub-id pub-id-type="medline">42091800</pub-id></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lorusso</surname><given-names>G</given-names> </name><name name-style="western"><surname>Ruscino</surname><given-names>G</given-names> </name><name name-style="western"><surname>Spitaleri</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Comparative evaluation of large language models for generating CAD-RADS 2.0-compliant diagnostic conclusions in cardiac CT reports</article-title><source>Insights Imaging</source><year>2026</year><month>04</month><day>22</day><volume>17</volume><issue>1</issue><fpage>112</fpage><pub-id pub-id-type="doi">10.1186/s13244-026-02285-6</pub-id><pub-id pub-id-type="medline">42018072</pub-id></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dong</surname><given-names>F</given-names> </name><name name-style="western"><surname>Nie</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>M</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Q</given-names> </name></person-group><article-title>Keyword-based AI assistance in the generation of radiology reports: a pilot study</article-title><source>NPJ Digit Med</source><year>2025</year><month>08</month><day>1</day><volume>8</volume><issue>1</issue><fpage>490</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01889-4</pub-id><pub-id pub-id-type="medline">40750683</pub-id></nlm-citation></ref><ref id="ref64"><label>64</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pellegrini</surname><given-names>C</given-names> </name><name name-style="western"><surname>&#x00D6;zsoy</surname><given-names>E</given-names> </name><name name-style="western"><surname>Gassert</surname><given-names>FT</given-names> </name><etal/></person-group><article-title>Enhancing radiology workflows through collaborative AI-assisted chest X-ray reporting using large vision-language models: a proof-of-concept study</article-title><source>Insights Imaging</source><year>2026</year><month>04</month><day>28</day><volume>17</volume><issue>1</issue><fpage>123</fpage><pub-id pub-id-type="doi">10.1186/s13244-026-02292-7</pub-id><pub-id pub-id-type="medline">42047956</pub-id></nlm-citation></ref><ref id="ref65"><label>65</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tan</surname><given-names>N</given-names> </name></person-group><article-title>Large language model&#x2013;assisted radiology reporting in a single-radiologist implementation: a retrospective cohort study interpreted through a UTAUT lens</article-title><source>Abdom Radiol</source><year>2026</year><pub-id pub-id-type="doi">10.1007/s00261-026-05524-y</pub-id></nlm-citation></ref><ref id="ref66"><label>66</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Guo</surname><given-names>R</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>P</given-names> </name><name name-style="western"><surname>Qian</surname><given-names>L</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>X</given-names> </name></person-group><article-title>Enhancing diagnostic accuracy and efficiency with GPT-4-generated structured reports: a comprehensive study</article-title><source>J Med Biol Eng</source><year>2024</year><month>02</month><volume>44</volume><issue>1</issue><fpage>144</fpage><lpage>153</lpage><pub-id pub-id-type="doi">10.1007/s40846-024-00849-9</pub-id></nlm-citation></ref><ref id="ref67"><label>67</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hong</surname><given-names>EK</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>S</given-names> </name><name name-style="western"><surname>Song</surname><given-names>O</given-names> </name><name name-style="western"><surname>Leung</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hammer</surname><given-names>M</given-names> </name><name name-style="western"><surname>Suh</surname><given-names>CH</given-names> </name></person-group><article-title>Temperature setting of a multimodal generative artificial intelligence model: association with accuracy and quality of artificial intelligence-generated chest radiograph reports</article-title><source>AJR Am J Roentgenol</source><year>2026</year><month>02</month><volume>226</volume><issue>2</issue><fpage>e2533704</fpage><pub-id pub-id-type="doi">10.2214/AJR.25.33704</pub-id><pub-id pub-id-type="medline">41159788</pub-id></nlm-citation></ref><ref id="ref68"><label>68</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zanardo</surname><given-names>M</given-names> </name><name name-style="western"><surname>Albano</surname><given-names>D</given-names> </name><name name-style="western"><surname>Molinari</surname><given-names>V</given-names> </name><etal/></person-group><article-title>Can AI write reports like a radiologist? A blinded evaluation of large language model-generated lumbar spine MRI reports</article-title><source>Eur Radiol Exp</source><year>2026</year><month>02</month><day>23</day><volume>10</volume><issue>1</issue><fpage>16</fpage><pub-id pub-id-type="doi">10.1186/s41747-026-00682-6</pub-id><pub-id pub-id-type="medline">41729370</pub-id></nlm-citation></ref><ref id="ref69"><label>69</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xie</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Tao</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Large language models for efficient whole-organ MRI score-based reports and categorization in knee osteoarthritis</article-title><source>Insights Imaging</source><year>2025</year><month>05</month><day>14</day><volume>16</volume><issue>1</issue><fpage>100</fpage><pub-id pub-id-type="doi">10.1186/s13244-025-01976-w</pub-id><pub-id pub-id-type="medline">40366500</pub-id></nlm-citation></ref><ref id="ref70"><label>70</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wo&#x017A;nicki</surname><given-names>P</given-names> </name><name name-style="western"><surname>Laqua</surname><given-names>C</given-names> </name><name name-style="western"><surname>Fiku</surname><given-names>I</given-names> </name><etal/></person-group><article-title>Automatic structuring of radiology reports with on-premise open-source large language models</article-title><source>Eur Radiol</source><year>2025</year><month>04</month><volume>35</volume><issue>4</issue><fpage>2018</fpage><lpage>2029</lpage><pub-id pub-id-type="doi">10.1007/s00330-024-11074-y</pub-id><pub-id pub-id-type="medline">39390261</pub-id></nlm-citation></ref><ref id="ref71"><label>71</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shao</surname><given-names>A</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>W</given-names> </name><etal/></person-group><article-title>Generative artificial intelligence for fundus fluorescein angiography interpretation and human expert evaluation</article-title><source>NPJ Digit Med</source><year>2025</year><month>07</month><day>2</day><volume>8</volume><issue>1</issue><fpage>396</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01759-z</pub-id><pub-id pub-id-type="medline">40603524</pub-id></nlm-citation></ref><ref id="ref72"><label>72</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Voinea</surname><given-names>&#x0218;V</given-names> </name><name name-style="western"><surname>M&#x0103;muleanu</surname><given-names>M</given-names> </name><name name-style="western"><surname>Teic&#x0103;</surname><given-names>RV</given-names> </name><name name-style="western"><surname>Florescu</surname><given-names>LM</given-names> </name><name name-style="western"><surname>Seli&#x0219;teanu</surname><given-names>D</given-names> </name><name name-style="western"><surname>Gheonea</surname><given-names>IA</given-names> </name></person-group><article-title>GPT-driven radiology report generation with fine-tuned Llama 3</article-title><source>Bioengineering (Basel)</source><year>2024</year><month>10</month><day>18</day><volume>11</volume><issue>10</issue><fpage>1043</fpage><pub-id pub-id-type="doi">10.3390/bioengineering11101043</pub-id><pub-id pub-id-type="medline">39451418</pub-id></nlm-citation></ref><ref id="ref73"><label>73</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Park</surname><given-names>J</given-names> </name><name name-style="western"><surname>Han</surname><given-names>K</given-names> </name><name name-style="western"><surname>Oh</surname><given-names>JS</given-names> </name><etal/></person-group><article-title>Prospective evaluation of artificial intelligence (AI) in lumbar spine magnetic resonance imaging (MRI) workflow: from deep learning (DL)&#x2013;enhanced accelerated acquisition to simultaneous vision language model (VLM)&#x2013;based automated report generation</article-title><source>Eur J Radiol</source><year>2026</year><month>03</month><volume>196</volume><fpage>112695</fpage><pub-id pub-id-type="doi">10.1016/j.ejrad.2026.112695</pub-id><pub-id pub-id-type="medline">41579672</pub-id></nlm-citation></ref><ref id="ref74"><label>74</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Soleimani</surname><given-names>M</given-names> </name><name name-style="western"><surname>Seyyedi</surname><given-names>N</given-names> </name><name name-style="western"><surname>Ayyoubzadeh</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Kalhori</surname><given-names>SRN</given-names> </name><name name-style="western"><surname>Keshavarz</surname><given-names>H</given-names> </name></person-group><article-title>Practical evaluation of ChatGPT performance for radiology report generation</article-title><source>Acad Radiol</source><year>2024</year><month>12</month><volume>31</volume><issue>12</issue><fpage>4823</fpage><lpage>4832</lpage><pub-id pub-id-type="doi">10.1016/j.acra.2024.07.020</pub-id><pub-id pub-id-type="medline">39142976</pub-id></nlm-citation></ref><ref id="ref75"><label>75</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Azhar</surname><given-names>K</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>BD</given-names> </name><name name-style="western"><surname>Byon</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>S</given-names> </name><name name-style="western"><surname>Cho</surname><given-names>KR</given-names> </name><name name-style="western"><surname>Song</surname><given-names>SE</given-names> </name></person-group><article-title>Semiautomated breast ultrasound report generation using multimodal large language models and deep learning</article-title><source>Front Med (Lausanne)</source><year>2026</year><volume>13</volume><fpage>1679203</fpage><pub-id pub-id-type="doi">10.3389/fmed.2026.1679203</pub-id><pub-id pub-id-type="medline">41647521</pub-id></nlm-citation></ref><ref id="ref76"><label>76</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>K</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>W</given-names> </name><name name-style="western"><surname>Li</surname><given-names>X</given-names> </name></person-group><article-title>The potential of Gemini and GPTs for structured report generation based on free-text <sup>18</sup>F-FDG PET/CT breast cancer reports</article-title><source>Acad Radiol</source><year>2025</year><month>02</month><volume>32</volume><issue>2</issue><fpage>624</fpage><lpage>633</lpage><pub-id pub-id-type="doi">10.1016/j.acra.2024.08.052</pub-id><pub-id pub-id-type="medline">39245597</pub-id></nlm-citation></ref><ref id="ref77"><label>77</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Raminedi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Shridevi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Won</surname><given-names>D</given-names> </name></person-group><article-title>Multi-modal transformer architecture for medical image analysis and automated report generation</article-title><source>Sci Rep</source><year>2024</year><month>08</month><day>20</day><volume>14</volume><issue>1</issue><fpage>19281</fpage><pub-id pub-id-type="doi">10.1038/s41598-024-69981-5</pub-id><pub-id pub-id-type="medline">39164302</pub-id></nlm-citation></ref><ref id="ref78"><label>78</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bosbach</surname><given-names>WA</given-names> </name><name name-style="western"><surname>Heide</surname><given-names>MS</given-names> </name><name name-style="western"><surname>G&#x00F6;zl&#x00FC;g&#x00F6;l</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Conceptual proposal for LLM-generated FDG PET/CT follow-up reports in melanoma: a pilot study on model stability and blinded expert evaluation</article-title><source>Front Nucl Med</source><year>2026</year><volume>6</volume><fpage>1723650</fpage><pub-id pub-id-type="doi">10.3389/fnume.2026.1723650</pub-id><pub-id pub-id-type="medline">41909529</pub-id></nlm-citation></ref><ref id="ref79"><label>79</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tran</surname><given-names>M</given-names> </name><name name-style="western"><surname>Schmidle</surname><given-names>P</given-names> </name><name name-style="western"><surname>Guo</surname><given-names>RR</given-names> </name><etal/></person-group><article-title>Generating dermatopathology reports from gigapixel whole slide images with HistoGPT</article-title><source>Nat Commun</source><year>2025</year><month>05</month><day>27</day><volume>16</volume><issue>1</issue><fpage>4886</fpage><pub-id pub-id-type="doi">10.1038/s41467-025-60014-x</pub-id><pub-id pub-id-type="medline">40419470</pub-id></nlm-citation></ref><ref id="ref80"><label>80</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>D</given-names> </name><name name-style="western"><surname>Li</surname><given-names>W</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>N</given-names> </name><etal/></person-group><article-title>SpineVLM: a markdown-guided structured fine-tuning framework for spine X-ray report generation</article-title><source>IEEE J Biomed Health Inform</source><year>2026</year><month>05</month><day>4</day><pub-id pub-id-type="doi">10.1109/JBHI.2026.3689568</pub-id><pub-id pub-id-type="medline">42081407</pub-id></nlm-citation></ref><ref id="ref81"><label>81</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ye</surname><given-names>X</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Q</given-names> </name><etal/></person-group><article-title>Report generation system for slit-lamp image interpretation using vision-language models</article-title><source>Ophthalmol Ther</source><year>2026</year><month>04</month><volume>15</volume><issue>4</issue><fpage>1509</fpage><lpage>1522</lpage><pub-id pub-id-type="doi">10.1007/s40123-026-01352-x</pub-id><pub-id pub-id-type="medline">41831132</pub-id></nlm-citation></ref><ref id="ref82"><label>82</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Barzegar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Rostami</surname><given-names>H</given-names> </name><name name-style="western"><surname>Sanati</surname><given-names>A</given-names> </name><name name-style="western"><surname>Afshoon</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kosari</surname><given-names>A</given-names> </name><name name-style="western"><surname>Keshavarz</surname><given-names>A</given-names> </name></person-group><article-title>Enhancing chest X-ray report generation with pathology-guided prompts and vision-language shortcut bias mitigation</article-title><source>Biomed Signal Process Control</source><year>2026</year><month>07</month><volume>120</volume><fpage>110173</fpage><pub-id pub-id-type="doi">10.1016/j.bspc.2026.110173</pub-id></nlm-citation></ref><ref id="ref83"><label>83</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Allaberdiev</surname><given-names>S</given-names> </name><name name-style="western"><surname>Khan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mamarasulov</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>X</given-names> </name></person-group><article-title>Chestxgen: dynamic memory&#x2013;augmented vision-language transformer with context-aware gating for radiology report generation</article-title><source>J Artif Intell Soft Comput Res</source><year>2026</year><month>01</month><day>1</day><volume>16</volume><issue>1</issue><fpage>55</fpage><lpage>72</lpage><pub-id pub-id-type="doi">10.2478/jaiscr-2026-0003</pub-id></nlm-citation></ref><ref id="ref84"><label>84</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>KH</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>CP</given-names> </name><name name-style="western"><surname>Kuo</surname><given-names>CT</given-names> </name><etal/></person-group><article-title>Automatic speech recognition and large language models for multilingual pathology report generation: proof-of-concept study</article-title><source>JMIR Form Res</source><year>2026</year><month>05</month><day>13</day><volume>10</volume><fpage>e90814</fpage><pub-id pub-id-type="doi">10.2196/90814</pub-id><pub-id pub-id-type="medline">42127277</pub-id></nlm-citation></ref><ref id="ref85"><label>85</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Khatoon</surname><given-names>S</given-names> </name><name name-style="western"><surname>Mahmood</surname><given-names>A</given-names> </name></person-group><article-title>Automated radiological report generation from breast ultrasound images using vision and language transformers</article-title><source>J Imaging</source><year>2026</year><month>02</month><day>6</day><volume>12</volume><issue>2</issue><fpage>68</fpage><pub-id pub-id-type="doi">10.3390/jimaging12020068</pub-id><pub-id pub-id-type="medline">41745433</pub-id></nlm-citation></ref><ref id="ref86"><label>86</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Park</surname><given-names>YW</given-names> </name><name name-style="western"><surname>Kang</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ryu</surname><given-names>H</given-names> </name><etal/></person-group><article-title>A robust vision language model for molecular status prediction and radiology report generation in adult-type diffuse gliomas</article-title><source>NPJ Digit Med</source><year>2026</year><month>04</month><day>2</day><volume>9</volume><issue>1</issue><fpage>406</fpage><pub-id pub-id-type="doi">10.1038/s41746-026-02581-x</pub-id><pub-id pub-id-type="medline">41927936</pub-id></nlm-citation></ref><ref id="ref87"><label>87</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hossain</surname><given-names>MZ</given-names> </name><name name-style="western"><surname>Ahmed</surname><given-names>M</given-names> </name><name name-style="western"><surname>Samu</surname><given-names>MSS</given-names> </name><name name-style="western"><surname>Islam</surname><given-names>MR</given-names> </name></person-group><article-title>Privacy-preserving chest X-ray report generation via multimodal federated learning with ViT and GPT2</article-title><source>Biomed Mater Devices</source><year>2025</year><pub-id pub-id-type="doi">10.1007/s44174-025-00538-4</pub-id></nlm-citation></ref><ref id="ref88"><label>88</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hosseini</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ibrahim</surname><given-names>A</given-names> </name><name name-style="western"><surname>Serag</surname><given-names>A</given-names> </name></person-group><article-title>M3: multimodal artificial intelligence for medical report generation and visual question answering from 3D abdominal CT scans</article-title><source>BJR Artif Intell</source><year>2025</year><volume>2</volume><issue>1</issue><fpage>ubaf011</fpage><pub-id pub-id-type="doi">10.1093/bjrai/ubaf011</pub-id><pub-id pub-id-type="medline">42063998</pub-id></nlm-citation></ref><ref id="ref89"><label>89</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>M</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Constructing a large language model to generate impressions from findings in radiology reports</article-title><source>Radiology</source><year>2024</year><month>09</month><volume>312</volume><issue>3</issue><fpage>e240885</fpage><pub-id pub-id-type="doi">10.1148/radiol.240885</pub-id><pub-id pub-id-type="medline">39287525</pub-id></nlm-citation></ref><ref id="ref90"><label>90</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sun</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Ong</surname><given-names>H</given-names> </name><name name-style="western"><surname>Kennedy</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Evaluating GPT-4 on impressions generation in radiology reports</article-title><source>Radiology</source><year>2023</year><month>06</month><volume>307</volume><issue>5</issue><fpage>e231259</fpage><pub-id pub-id-type="doi">10.1148/radiol.231259</pub-id><pub-id pub-id-type="medline">37367439</pub-id></nlm-citation></ref><ref id="ref91"><label>91</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Neill</surname><given-names>L</given-names> </name><name name-style="western"><surname>Wittbrodt</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Generative artificial intelligence for chest radiograph interpretation in the emergency department</article-title><source>JAMA Netw Open</source><year>2023</year><month>10</month><day>2</day><volume>6</volume><issue>10</issue><fpage>e2336100</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2023.36100</pub-id><pub-id pub-id-type="medline">37796505</pub-id></nlm-citation></ref><ref id="ref92"><label>92</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mallio</surname><given-names>CA</given-names> </name><name name-style="western"><surname>Bernetti</surname><given-names>C</given-names> </name><name name-style="western"><surname>Sertorio</surname><given-names>AC</given-names> </name><name name-style="western"><surname>Zobel</surname><given-names>BB</given-names> </name></person-group><article-title>ChatGPT in radiology structured reporting: analysis of ChatGPT-3.5 Turbo and GPT-4 in reducing word count and recalling findings</article-title><source>Quant Imaging Med Surg</source><year>2024</year><month>02</month><day>1</day><volume>14</volume><issue>2</issue><fpage>2096</fpage><lpage>2102</lpage><pub-id pub-id-type="doi">10.21037/qims-23-1300</pub-id><pub-id pub-id-type="medline">38415145</pub-id></nlm-citation></ref><ref id="ref93"><label>93</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Stephan</surname><given-names>D</given-names> </name><name name-style="western"><surname>Bertsch</surname><given-names>A</given-names> </name><name name-style="western"><surname>Burwinkel</surname><given-names>M</given-names> </name><etal/></person-group><article-title>AI in dental radiology improving the efficiency of reporting with ChatGPT: comparative study</article-title><source>J Med Internet Res</source><year>2024</year><month>12</month><day>23</day><volume>26</volume><fpage>e60684</fpage><pub-id pub-id-type="doi">10.2196/60684</pub-id><pub-id pub-id-type="medline">39714078</pub-id></nlm-citation></ref><ref id="ref94"><label>94</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>C</given-names> </name><name name-style="western"><surname>Cho</surname><given-names>S</given-names> </name><name name-style="western"><surname>Yoon</surname><given-names>JH</given-names> </name></person-group><article-title>Utility of multimodal large language models in analyzing chest X-rays with incomplete contextual information</article-title><source>Healthc Inform Res</source><year>2025</year><month>10</month><volume>31</volume><issue>4</issue><fpage>416</fpage><lpage>425</lpage><pub-id pub-id-type="doi">10.4258/hir.2025.31.4.416</pub-id><pub-id pub-id-type="medline">41265427</pub-id></nlm-citation></ref><ref id="ref95"><label>95</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Di Palma</surname><given-names>L</given-names> </name><name name-style="western"><surname>Darvizeh</surname><given-names>F</given-names> </name><name name-style="western"><surname>Al&#x00EC;</surname><given-names>M</given-names> </name><name name-style="western"><surname>Fazzini</surname><given-names>D</given-names> </name></person-group><article-title>Structured transformation of unstructured prostate MRI reports using large language models</article-title><source>Tomography</source><year>2025</year><month>06</month><day>17</day><volume>11</volume><issue>6</issue><fpage>69</fpage><pub-id pub-id-type="doi">10.3390/tomography11060069</pub-id><pub-id pub-id-type="medline">40560015</pub-id></nlm-citation></ref><ref id="ref96"><label>96</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gupta</surname><given-names>A</given-names> </name><name name-style="western"><surname>Malhotra</surname><given-names>H</given-names> </name><name name-style="western"><surname>Garg</surname><given-names>AK</given-names> </name><name name-style="western"><surname>Rangarajan</surname><given-names>K</given-names> </name></person-group><article-title>Enhancing radiological reporting in head and neck cancer: converting free-text CT scan reports to structured reports using large language models</article-title><source>Indian J Radiol Imaging</source><year>2025</year><volume>35</volume><issue>1</issue><fpage>43</fpage><lpage>49</lpage><pub-id pub-id-type="doi">10.1055/s-0044-1788589</pub-id><pub-id pub-id-type="medline">39697521</pub-id></nlm-citation></ref><ref id="ref97"><label>97</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>S</given-names> </name><name name-style="western"><surname>Youn</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>H</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>M</given-names> </name><name name-style="western"><surname>Yoon</surname><given-names>SH</given-names> </name></person-group><article-title>CXR-LLaVA: a multimodal large language model for interpreting chest X-ray images</article-title><source>Eur Radiol</source><year>2025</year><volume>35</volume><issue>7</issue><fpage>4374</fpage><lpage>4386</lpage><pub-id pub-id-type="doi">10.1007/s00330-024-11339-6</pub-id></nlm-citation></ref><ref id="ref98"><label>98</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pesapane</surname><given-names>F</given-names> </name><name name-style="western"><surname>Nicosia</surname><given-names>L</given-names> </name><name name-style="western"><surname>Rotili</surname><given-names>A</given-names> </name><etal/></person-group><article-title>A preliminary investigation into the potential, pitfalls, and limitations of large language models for mammography interpretation</article-title><source>Discov Oncol</source><year>2025</year><month>02</month><day>24</day><volume>16</volume><issue>1</issue><fpage>233</fpage><pub-id pub-id-type="doi">10.1007/s12672-025-02005-4</pub-id><pub-id pub-id-type="medline">39992569</pub-id></nlm-citation></ref><ref id="ref99"><label>99</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hartsock</surname><given-names>I</given-names> </name><name name-style="western"><surname>Araujo</surname><given-names>C</given-names> </name><name name-style="western"><surname>Folio</surname><given-names>L</given-names> </name><name name-style="western"><surname>Rasool</surname><given-names>G</given-names> </name></person-group><article-title>Improving radiology report conciseness and structure via local large language models</article-title><source>J Imaging Inform Med</source><year>2026</year><volume>39</volume><issue>1</issue><fpage>1005</fpage><lpage>1016</lpage><pub-id pub-id-type="doi">10.1007/s10278-025-01510-w</pub-id><pub-id pub-id-type="medline">40259201</pub-id></nlm-citation></ref><ref id="ref100"><label>100</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Silbergleit</surname><given-names>M</given-names> </name><name name-style="western"><surname>T&#x00F3;th</surname><given-names>A</given-names> </name><name name-style="western"><surname>Chamberlin</surname><given-names>JH</given-names> </name><etal/></person-group><article-title>ChatGPT vs Gemini: comparative accuracy and efficiency in CAD-RADS score assignment from radiology reports</article-title><source>J Imaging Inform Med</source><year>2025</year><month>08</month><volume>38</volume><issue>4</issue><fpage>2303</fpage><lpage>2311</lpage><pub-id pub-id-type="doi">10.1007/s10278-024-01328-y</pub-id><pub-id pub-id-type="medline">39528887</pub-id></nlm-citation></ref><ref id="ref101"><label>101</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Adams</surname><given-names>LC</given-names> </name><name name-style="western"><surname>Truhn</surname><given-names>D</given-names> </name><name name-style="western"><surname>Busch</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Leveraging GPT-4 for post hoc transformation of free-text radiology reports into structured reporting: a multilingual feasibility study</article-title><source>Radiology</source><year>2023</year><month>05</month><volume>307</volume><issue>4</issue><fpage>e230725</fpage><pub-id pub-id-type="doi">10.1148/radiol.230725</pub-id><pub-id pub-id-type="medline">37014240</pub-id></nlm-citation></ref><ref id="ref102"><label>102</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jiang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Xia</surname><given-names>S</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Transforming free-text radiology reports into structured reports using ChatGPT: a study on thyroid ultrasonography</article-title><source>Eur J Radiol</source><year>2024</year><month>06</month><volume>175</volume><fpage>111458</fpage><pub-id pub-id-type="doi">10.1016/j.ejrad.2024.111458</pub-id><pub-id pub-id-type="medline">38613868</pub-id></nlm-citation></ref><ref id="ref103"><label>103</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hasani</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zahergivar</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Evaluating the performance of generative pre-trained transformer-4 (GPT-4) in standardizing radiology reports</article-title><source>Eur Radiol</source><year>2024</year><month>06</month><volume>34</volume><issue>6</issue><fpage>3566</fpage><lpage>3574</lpage><pub-id pub-id-type="doi">10.1007/s00330-023-10384-x</pub-id><pub-id pub-id-type="medline">37938381</pub-id></nlm-citation></ref><ref id="ref104"><label>104</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tsaniya</surname><given-names>H</given-names> </name><name name-style="western"><surname>Fatichah</surname><given-names>C</given-names> </name><name name-style="western"><surname>Suciati</surname><given-names>N</given-names> </name><name name-style="western"><surname>Obi</surname><given-names>T</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>JS</given-names> </name></person-group><article-title>Medical report generation with knowledge distillation and multi-stage hierarchical attention in vision transformer encoder and GPT-2 decoder</article-title><source>IEEE Access</source><year>2025</year><volume>13</volume><fpage>132973</fpage><lpage>132989</lpage><pub-id pub-id-type="doi">10.1109/ACCESS.2025.3588344</pub-id></nlm-citation></ref><ref id="ref105"><label>105</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yan</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Han</surname><given-names>G</given-names> </name></person-group><article-title>Automated ultrasound diagnosis via CLIP-GPT synergy: a multimodal framework for image classification and report generation</article-title><source>IEEE Access</source><year>2025</year><volume>13</volume><fpage>107950</fpage><lpage>107960</lpage><pub-id pub-id-type="doi">10.1109/ACCESS.2025.3578462</pub-id></nlm-citation></ref><ref id="ref106"><label>106</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>L&#x00F3;pez-&#x00DA;beda</surname><given-names>P</given-names> </name><name name-style="western"><surname>Mart&#x00ED;n-Noguerol</surname><given-names>T</given-names> </name><name name-style="western"><surname>Escart&#x00ED;n</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cabrera-Zubizarreta</surname><given-names>A</given-names> </name><name name-style="western"><surname>Luna</surname><given-names>A</given-names> </name></person-group><article-title>Automated MRI pituitary structured reporting from free-text using a fine-tuned Llama model: a feasibility study</article-title><source>Jpn J Radiol</source><year>2025</year><month>05</month><volume>43</volume><issue>5</issue><fpage>770</fpage><lpage>778</lpage><pub-id pub-id-type="doi">10.1007/s11604-024-01721-1</pub-id><pub-id pub-id-type="medline">39730936</pub-id></nlm-citation></ref><ref id="ref107"><label>107</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tian</surname><given-names>W</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Cheng</surname><given-names>T</given-names> </name><etal/></person-group><article-title>A medical multimodal large language model for pediatric pneumonia</article-title><source>IEEE J Biomed Health Inform</source><year>2025</year><month>09</month><volume>29</volume><issue>9</issue><fpage>6869</fpage><lpage>6882</lpage><pub-id pub-id-type="doi">10.1109/JBHI.2025.3569361</pub-id><pub-id pub-id-type="medline">40354198</pub-id></nlm-citation></ref><ref id="ref108"><label>108</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Q</given-names> </name><etal/></person-group><article-title>Towards generalizable pathology reports via a multimodal LLM with the multicenter in-context learning</article-title><source>Med Image Anal</source><year>2026</year><month>06</month><volume>111</volume><fpage>104060</fpage><pub-id pub-id-type="doi">10.1016/j.media.2026.104060</pub-id></nlm-citation></ref><ref id="ref109"><label>109</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Garc&#x00ED;a Hidalgo</surname><given-names>C</given-names> </name><name name-style="western"><surname>Consentino Hern&#x00E1;ndez</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Cayuela Esp&#x00ED;</surname><given-names>JV</given-names> </name><etal/></person-group><article-title>Informe radiol&#x00F3;gico estructurado asistido por modelos de lenguaje en residentes de Radiolog&#x00ED;a: piloto de implementaci&#x00F3;n en Urgencias [Article in Spanish]</article-title><source>Rev Esp Edu Med</source><year>2026</year><volume>7</volume><issue>1</issue><pub-id pub-id-type="doi">10.6018/edumed.695571</pub-id></nlm-citation></ref><ref id="ref110"><label>110</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tasneem</surname><given-names>N</given-names> </name><name name-style="western"><surname>van der Pol</surname><given-names>CB</given-names> </name><name name-style="western"><surname>Zahoor</surname><given-names>A</given-names> </name><etal/></person-group><article-title>From radiology findings to artificial intelligence&#x2013;powered impressions: a retrospective study on the comparative performance of recent large language models</article-title><source>Intell Med</source><year>2026</year><month>04</month><volume>6</volume><issue>2</issue><fpage>154</fpage><lpage>165</lpage><pub-id pub-id-type="doi">10.1016/j.imed.2025.11.003</pub-id></nlm-citation></ref><ref id="ref111"><label>111</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sheng</surname><given-names>W</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Xiao</surname><given-names>L</given-names> </name><etal/></person-group><article-title>BI-RADS-compliant structured mammography reporting using locally deployed large language models under privacy constraints</article-title><source>Eur Radiol</source><year>2026</year><month>05</month><volume>36</volume><issue>5</issue><fpage>3676</fpage><lpage>3686</lpage><pub-id pub-id-type="doi">10.1007/s00330-025-12147-2</pub-id><pub-id pub-id-type="medline">41258455</pub-id></nlm-citation></ref><ref id="ref112"><label>112</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>R</given-names> </name><name name-style="western"><surname>Bai</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yue</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>P</given-names> </name></person-group><article-title>Teaching multimodal LLMs to comprehend 12-lead electrocardiographic images</article-title><source>NPJ Digit Med</source><year>2026</year><month>03</month><day>16</day><volume>9</volume><issue>1</issue><fpage>349</fpage><pub-id pub-id-type="doi">10.1038/s41746-026-02551-3</pub-id><pub-id pub-id-type="medline">41840182</pub-id></nlm-citation></ref><ref id="ref113"><label>113</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>C</given-names> </name><name name-style="western"><surname>Park</surname><given-names>S</given-names> </name><name name-style="western"><surname>Shin</surname><given-names>CI</given-names> </name><etal/></person-group><article-title>Read like a radiologist: efficient vision-language model for 3D medical imaging interpretation</article-title><source>Med Image Anal</source><year>2026</year><month>06</month><volume>111</volume><fpage>104077</fpage><pub-id pub-id-type="doi">10.1016/j.media.2026.104077</pub-id><pub-id pub-id-type="medline">41990528</pub-id></nlm-citation></ref><ref id="ref114"><label>114</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alkhaldi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Alnajim</surname><given-names>R</given-names> </name><name name-style="western"><surname>Alabdullatef</surname><given-names>L</given-names> </name><name name-style="western"><surname>Alyahya</surname><given-names>R</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>D</given-names> </name><etal/></person-group><article-title>MiniGPT-MED: a unified vision-language model for radiology image understanding</article-title><source>Trans Mach Learn Res</source><year>2026</year><access-date>2026-08-01</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://openreview.net/pdf?id=NenHFEg1Di">https://openreview.net/pdf?id=NenHFEg1Di</ext-link></comment></nlm-citation></ref><ref id="ref115"><label>115</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abdaoui</surname><given-names>H</given-names> </name><name name-style="western"><surname>Barbaria</surname><given-names>S</given-names> </name><name name-style="western"><surname>Dergaa</surname><given-names>I</given-names> </name><etal/></person-group><article-title>MedFusionT5: cross-modal attention boosts semantic quality and reduces hallucinations in dental AI</article-title><source>Int Dent J</source><year>2026</year><month>06</month><volume>76</volume><issue>3</issue><fpage>109404</fpage><pub-id pub-id-type="doi">10.1016/j.identj.2025.109404</pub-id><pub-id pub-id-type="medline">41771189</pub-id></nlm-citation></ref><ref id="ref116"><label>116</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>W</given-names> </name><name name-style="western"><surname>Qin</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>Y</given-names> </name></person-group><article-title>CT-Agent: a multimodal-LLM agent for 3D CT radiology question answering</article-title><source>Sci China Inf Sci</source><year>2026</year><month>05</month><volume>69</volume><issue>5</issue><fpage>150107</fpage><pub-id pub-id-type="doi">10.1007/s11432-025-4818-7</pub-id></nlm-citation></ref><ref id="ref117"><label>117</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rouhizadeh</surname><given-names>H</given-names> </name><name name-style="western"><surname>Sandralegar</surname><given-names>A</given-names> </name><name name-style="western"><surname>Yazdani</surname><given-names>A</given-names> </name><etal/></person-group><article-title>The detectability paradox: bilingual medical report generation with open-weight models and the limits of human oversight</article-title><source>J Am Med Inform Assoc</source><year>2026</year><month>07</month><day>1</day><volume>33</volume><issue>7</issue><fpage>1303</fpage><lpage>1313</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocag070</pub-id><pub-id pub-id-type="medline">42097830</pub-id></nlm-citation></ref><ref id="ref118"><label>118</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Pan</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhong</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Potential of multimodal large language models for data mining of medical images and free-text reports</article-title><source>Meta-Radiology</source><year>2024</year><month>12</month><volume>2</volume><issue>4</issue><fpage>100103</fpage><pub-id pub-id-type="doi">10.1016/j.metrad.2024.100103</pub-id></nlm-citation></ref><ref id="ref119"><label>119</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bosbach</surname><given-names>WA</given-names> </name><name name-style="western"><surname>Senge</surname><given-names>JF</given-names> </name><name name-style="western"><surname>Nemeth</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Ability of ChatGPT to generate competent radiology reports for distal radius fracture by use of RSNA template items and integrated AO classifier</article-title><source>Curr Probl Diagn Radiol</source><year>2024</year><volume>53</volume><issue>1</issue><fpage>102</fpage><lpage>110</lpage><pub-id pub-id-type="doi">10.1067/j.cpradiol.2023.04.001</pub-id><pub-id pub-id-type="medline">37263804</pub-id></nlm-citation></ref><ref id="ref120"><label>120</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Xing</surname><given-names>S</given-names> </name><name name-style="western"><surname>Li</surname><given-names>K</given-names> </name><name name-style="western"><surname>Guo</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Li</surname><given-names>G</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>C</given-names> </name></person-group><article-title>Automated generation of chest X-ray imaging diagnostic reports by multimodal and multi granularity features fusion</article-title><source>Biomed Signal Process Control</source><year>2025</year><month>07</month><volume>105</volume><fpage>107562</fpage><pub-id pub-id-type="doi">10.1016/j.bspc.2025.107562</pub-id></nlm-citation></ref><ref id="ref121"><label>121</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Edirisinghe</surname><given-names>D</given-names> </name><name name-style="western"><surname>Nimalsiri</surname><given-names>W</given-names> </name><name name-style="western"><surname>Hennayake</surname><given-names>M</given-names> </name><name name-style="western"><surname>Meedeniya</surname><given-names>D</given-names> </name><name name-style="western"><surname>Lim</surname><given-names>G</given-names> </name></person-group><article-title>Chest X-ray report generation using abnormality guided vision language model</article-title><source>IEEE Access</source><year>2025</year><volume>13</volume><fpage>157651</fpage><lpage>157673</lpage><pub-id pub-id-type="doi">10.1109/ACCESS.2025.3606961</pub-id></nlm-citation></ref><ref id="ref122"><label>122</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>X</given-names> </name><name name-style="western"><surname>He</surname><given-names>H</given-names> </name><name name-style="western"><surname>Feng</surname><given-names>J</given-names> </name></person-group><article-title>Context-enhanced framework for medical image report generation using multimodal contexts</article-title><source>Knowl Based Syst</source><year>2025</year><month>02</month><volume>310</volume><fpage>112913</fpage><pub-id pub-id-type="doi">10.1016/j.knosys.2024.112913</pub-id></nlm-citation></ref><ref id="ref123"><label>123</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Hong</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>DC-RRG: diagnosis-centered cascaded radiology report generation</article-title><source>Expert Syst Appl</source><year>2026</year><month>03</month><volume>299</volume><fpage>129884</fpage><pub-id pub-id-type="doi">10.1016/j.eswa.2025.129884</pub-id></nlm-citation></ref><ref id="ref124"><label>124</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Izhar</surname><given-names>A</given-names> </name><name name-style="western"><surname>Idris</surname><given-names>N</given-names> </name><name name-style="western"><surname>Japar</surname><given-names>N</given-names> </name></person-group><article-title>Engaging preference optimization alignment in large language model for continual radiology report generation: a hybrid approach</article-title><source>Cogn Comput</source><year>2025</year><volume>17</volume><issue>1</issue><pub-id pub-id-type="doi">10.1007/s12559-025-10404-6</pub-id></nlm-citation></ref><ref id="ref125"><label>125</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Leonardi</surname><given-names>G</given-names> </name><name name-style="western"><surname>Portinale</surname><given-names>L</given-names> </name><name name-style="western"><surname>Santomauro</surname><given-names>A</given-names> </name></person-group><article-title>Enhancing radiology report generation through pre-trained language models</article-title><source>Prog Artif Intell</source><year>2024</year><pub-id pub-id-type="doi">10.1007/s13748-024-00358-5</pub-id></nlm-citation></ref><ref id="ref126"><label>126</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aslam</surname><given-names>SM</given-names> </name></person-group><article-title>Generative artificial intelligence for the automated generation of medical reports: a modified poor and rich optimization (MPRO)&#x2013;based approach</article-title><source>Traitement du Signal</source><year>2025</year><month>02</month><day>28</day><volume>42</volume><issue>1</issue><fpage>409</fpage><lpage>422</lpage><pub-id pub-id-type="doi">10.18280/ts.420135</pub-id></nlm-citation></ref><ref id="ref127"><label>127</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Deng</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Grounded knowledge&#x2013;enhanced medical vision&#x2013;language pre-training for chest X-ray</article-title><source>Biomed Signal Process Control</source><year>2026</year><month>02</month><volume>112</volume><fpage>108416</fpage><pub-id pub-id-type="doi">10.1016/j.bspc.2025.108416</pub-id></nlm-citation></ref><ref id="ref128"><label>128</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>S-RRG-Bench: structured radiology report generation with fine-grained evaluation framework</article-title><source>Meta-Radiology</source><year>2025</year><month>12</month><volume>3</volume><issue>4</issue><fpage>100171</fpage><pub-id pub-id-type="doi">10.1016/j.metrad.2025.100171</pub-id></nlm-citation></ref><ref id="ref129"><label>129</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ucan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kaya</surname><given-names>B</given-names> </name><name name-style="western"><surname>Kaya</surname><given-names>M</given-names> </name></person-group><article-title>Turkish chest X-ray report generation model using the Swin enhanced yield transformer (Model-SEY) framework</article-title><source>Diagnostics (Basel)</source><year>2025</year><month>07</month><day>17</day><volume>15</volume><issue>14</issue><fpage>1805</fpage><pub-id pub-id-type="doi">10.3390/diagnostics15141805</pub-id><pub-id pub-id-type="medline">40722555</pub-id></nlm-citation></ref><ref id="ref130"><label>130</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rakibul Islam</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zahid Hossain</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ahmed</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sharmin Sultana Samu</surname><given-names>M</given-names> </name></person-group><article-title>Vision-language models for automated chest X-ray interpretation: leveraging ViT and GPT-2</article-title><source>Eng Rep</source><year>2025</year><month>06</month><volume>7</volume><issue>6</issue><pub-id pub-id-type="doi">10.1002/eng2.70220</pub-id></nlm-citation></ref><ref id="ref131"><label>131</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vasey</surname><given-names>B</given-names> </name><name name-style="western"><surname>Nagendran</surname><given-names>M</given-names> </name><name name-style="western"><surname>Campbell</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Reporting guideline for the early-stage clinical evaluation of decision support systems driven by artificial intelligence: DECIDE-AI</article-title><source>Nat Med</source><year>2022</year><month>05</month><volume>28</volume><issue>5</issue><fpage>924</fpage><lpage>933</lpage><pub-id pub-id-type="doi">10.1038/s41591-022-01772-9</pub-id><pub-id pub-id-type="medline">35585198</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Exact search strategies and search update summary.</p><media xlink:href="jmir_v28i1e97007_app1.docx" xlink:title="DOCX File, 41 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Supplementary evidence tables.</p><media xlink:href="jmir_v28i1e97007_app2.docx" xlink:title="DOCX File, 75 KB"/></supplementary-material><supplementary-material id="app3"><label>Checklist 1</label><p>Reporting guideline checklists: PRISMA 2020, PRISMA abstracts, PRISMA-S, and SWiM.</p><media xlink:href="jmir_v28i1e97007_app3.docx" xlink:title="DOCX File, 37 KB"/></supplementary-material></app-group></back></article>