<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="review-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e104092</article-id><article-id pub-id-type="doi">10.2196/104092</article-id><article-categories><subj-group subj-group-type="heading"><subject>Review</subject></subj-group></article-categories><title-group><article-title>Fine-Tuning, Retrieval-Augmented Generation, and Hybrid Adaptation of Language Models for Clinical Decision-Making in Health Care: Systematic Review</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Patel</surname><given-names>Anshum</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Khand</surname><given-names>Yugant</given-names></name><degrees>MBBS</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Vallamchetla</surname><given-names>Sai Krishna</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Li</surname><given-names>Pengze</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Tao</surname><given-names>Cui</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Cheung</surname><given-names>Joseph</given-names></name><degrees>MD, MS</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff3">3</xref></contrib></contrib-group><aff id="aff1"><institution>Division of Pulmonary, Allergy and Sleep Medicine, Mayo Clinic</institution><addr-line>4500 San Pablo Road South</addr-line><addr-line>Jacksonville</addr-line><addr-line>FL</addr-line><country>United States</country></aff><aff id="aff2"><institution>Department of Neurology, Mayo Clinic in Florida</institution><addr-line>Jacksonville</addr-line><addr-line>FL</addr-line><country>United States</country></aff><aff id="aff3"><institution>Department of Artificial Intelligence and Informatics, Mayo Clinic in Florida</institution><addr-line>Jacksonville</addr-line><addr-line>FL</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Beunza</surname><given-names>Juan-Jose</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Honarmand</surname><given-names>Mohammadmahdi</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Bhetwal</surname><given-names>Sagar</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Joseph Cheung, MD, MS, Division of Pulmonary, Allergy and Sleep Medicine, Mayo Clinic, 4500 San Pablo Road South, Jacksonville, FL, United States, 1 904-953-2000; <email>Cheung.Joseph@mayo.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>17</day><month>9</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e104092</elocation-id><history><date date-type="received"><day>08</day><month>06</month><year>2026</year></date><date date-type="rev-recd"><day>23</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>24</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Anshum Patel, Yugant Khand, Sai Krishna Vallamchetla, Pengze Li, Cui Tao, Joseph Cheung. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 17.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e104092"/><abstract><sec><title>Background</title><p>Large language models (LLMs) demonstrate strong performance on medical knowledge benchmarks, but their safe and effective use in clinical practice depends on posttraining adaptation rather than raw model capability. Fine-tuning, retrieval-augmented generation (RAG), and hybrid approaches are principal strategies for grounding language models in clinical evidence, yet their comparative effectiveness remains unclear.</p></sec><sec><title>Objective</title><p>This systematic review aims to synthesize evidence on fine-tuning, RAG, and hybrid posttraining strategies for clinical diagnosis and decision-support tasks and to identify strategy-task alignments and methodological features associated with improved performance.</p></sec><sec sec-type="methods"><title>Methods</title><p>We conducted a systematic review in accordance with PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) 2020 guidelines. PubMed/MEDLINE, Scopus, and Web of Science were searched from January 2018 through May 2026. Eligible studies evaluated transformer-based language models that underwent posttraining adaptation, retrieval augmentation, or both for clinical decision support, diagnosis, triage, risk stratification, or related health care applications. Studies evaluating nonadapted models, non&#x2013;language-model AI systems, prompt engineering without performance evaluation, or nonclinical applications were excluded. Data extracted included model architecture, adaptation strategy, clinical domain, validation approach, and performance outcomes. Risk of bias was assessed using PROBAST+AI (Prediction model Risk of Bias Assessment Tool for AI). Studies were grouped according to the primary enhancement strategy (fine-tuning or parameter-efficient fine-tuning, RAG, or hybrid approaches), and findings were synthesized descriptively.</p></sec><sec sec-type="results"><title>Results</title><p>Of 1890 identified records, 35 studies published between 2024 and 2026 met eligibility criteria. Enhancement strategies included RAG (17/35, 48.6%), fine-tuning or parameter-efficient fine-tuning (7/35, 20%), and hybrid approaches (11/35, 31.4%). Studies included diverse specialties from oncology, neurology, radiology, mental health, cardiology, ophthalmology, and surgical care. Fine-tuning demonstrated strong performance for task-specific applications, achieving area under the receiver operating characteristic curve values up to 0.912 for cancer detection and area under curve of 0.892 for major depressive disorder prediction, while matching clinician-level diagnostic performance in several studies. RAG improved guideline adherence and diagnostic accuracy, with increases from 71.1% to 92.1% and from 78.9% to 94.7% in guideline-based decision-support tasks. However, benefits were inconsistent across larger reasoning-capable models. Hybrid systems generally achieved the strongest performance in complex clinical workflows, with external validation accuracies exceeding 90% in stroke triage, dermatology, multimodal imaging, and oncology applications. Risk-of-bias assessment identified substantial methodological limitations, with 25 studies judged as high risk, 9 as unclear risk, and only 1 as low risk overall. Common concerns included inadequate external validation, lack of calibration assessment, nonrepresentative participant selection, and insufficient reporting of analytical methods.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Adaptation strategies should align with task needs, using fine-tuning for narrow classification, RAG for guideline-grounded reasoning, and hybrid approaches for complex multimodal tasks. However, the evidence base remains largely retrospective or benchmark-based. Prospective studies with external validation, calibration, and standardized safety reporting are needed before broader clinical use.</p></sec><sec><title>Trial Registration</title><p>PROSPERO CRD420261308522; https://www.crd.york.ac.uk/PROSPERO/view/CRD420261308522</p></sec></abstract><kwd-group><kwd>AI</kwd><kwd>language models</kwd><kwd>large language models</kwd><kwd>LLMs</kwd><kwd>small language models</kwd><kwd>SLMs</kwd><kwd>generative AI</kwd><kwd>fine-tuning</kwd><kwd>retrieval-augmented generation</kwd><kwd>RAG</kwd><kwd>reinforcement learning</kwd><kwd>natural language processing</kwd><kwd>clinical decision support</kwd><kwd>health care</kwd><kwd>AI agents</kwd><kwd>evidence-based medicine</kwd><kwd>posttraining methods</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Large language models (LLMs) are reshaping medical AI, but their clinical value depends less on raw benchmark performance than on how effectively they are adapted for real-world care. Medical AI has evolved from narrow systems built for single tasks, such as image classification or risk prediction, toward more general foundation models that can work across language, images, and structured clinical data [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref4">4</xref>]. Generative AI accelerated this shift by making language the main interface: LLMs can follow natural-language instructions, summarize records, answer questions, and draft communication in ways that resemble everyday clinical work more closely than earlier back-end prediction systems [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>]. This broader capability has made LLMs plausible tools for documentation, patient communication, knowledge synthesis, and diagnostic support, while also sharpening concerns about reliability, bias, transparency, and accountability [<xref ref-type="bibr" rid="ref4">4</xref>-<xref ref-type="bibr" rid="ref6">6</xref>].</p><p>Early medical language model research understandably focused on proof of capability. In MultiMedQA, Flan-PaLM (Google Research) achieved 67.6% accuracy on MedQA, surpassing the previous state of the art by more than 17 percentage points [<xref ref-type="bibr" rid="ref7">7</xref>]. Med-PaLM 2 (Google Research) later reached 86.5% on MedQA and, in pairwise evaluation of 1066 consumer medical questions, was preferred over physician answers across 8 of 9 clinically relevant cases [<xref ref-type="bibr" rid="ref7">7</xref>]. Signals of practical value also emerged in communication and summarization tasks: in a JAMA (Journal of the American Medical Association) Internal Medicine study of 195 patient questions, chatbot responses were preferred in 78.6% of evaluations and approximately 4 to 1 overall over physician responses [<xref ref-type="bibr" rid="ref8">8</xref>], and in a Nature Medicine reader study, the best-adapted LLM summaries were judged equivalent to medical experts in 45% of cases and superior in 36% of cases [<xref ref-type="bibr" rid="ref9">9</xref>]. Together, these studies established that LLMs can encode substantial clinical knowledge and perform well on selected language-heavy tasks, but they did so mainly under curated conditions [<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref10">10</xref>].</p><p>Those gains, however, should not be mistaken for clinical readiness. Examination questions and curated vignettes offer complete information and a constrained answer space, and therefore do not fully test uncertainty, evolving context, workflow integration, or downstream effects on clinician performance [<xref ref-type="bibr" rid="ref11">11</xref>-<xref ref-type="bibr" rid="ref13">13</xref>]. When evaluations moved closer to real diagnostic reasoning, results became more mixed. In a complex diagnostic challenge, GPT-4 (OpenAI) included the final diagnosis in its differential in 64% of cases and ranked it first in 39% of cases [<xref ref-type="bibr" rid="ref14">14</xref>]. In another physician-comparison study, GPT-4 achieved higher median Revised-IDEA (Interpretive Summary, Differential Diagnosis, Explanation of Reasoning, and Alternatives) reasoning scores than attendings and residents, but it also produced incorrect reasoning more often than residents [<xref ref-type="bibr" rid="ref15">15</xref>]. Most importantly, in a randomized clinical trial including 50 physicians, access to a commercial LLM did not significantly improve diagnostic reasoning compared with conventional resources alone [<xref ref-type="bibr" rid="ref16">16</xref>]. At the same time, current models remain vulnerable to hallucination, poor instruction following, sensitivity to the quantity and order of information, race-based medical content, and the introduction of false details, while narrow evaluations may miss clinically important equity harms [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>].</p><p>These limitations have shifted the field from asking whether language models can answer medical questions to how they should be improved for dependable clinical use. Fine-tuning aligns a base model with specialized terminology, documentation styles, and target tasks; in clinical summarization and medical evidence synthesis, adapted models have shown meaningful gains and can narrow the gap between open and proprietary systems [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref19">19</xref>]. Retrieval-augmented generation (RAG) addresses a different problem by grounding outputs in external knowledge at inference time, allowing systems to draw on guidelines, literature, or local protocols without retraining the underlying model [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref20">20</xref>]. This approach is especially attractive in medicine, where knowledge changes quickly and local context matters. In a radiology consultation study, adding RAG to a locally deployable model reduced hallucinations from 8% to 0% and improved mean response rank, while preserving the privacy advantages of local deployment [<xref ref-type="bibr" rid="ref21">21</xref>]. Increasingly, the most clinically plausible systems are hybrid, combining model adaptation, retrieval, and structured prompting rather than relying on any single method alone [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>].</p><p>Despite rapid progress, the evidence base for language model improvement methods remains fragmented by specialty, task, base model, data source, and evaluation design [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref12">12</xref>]. In a recent high-level review, 4609 peer-reviewed clinical language model studies were identified, yet only 1048 used real-world patient data and only 19 were prospective randomized trials [<xref ref-type="bibr" rid="ref11">11</xref>]. This gap is especially important for posttraining methods because choices about fine-tuning, retrieval, and hybrid design affect not only performance but also updatability, privacy, traceability, and fit within regulated clinical workflows. We therefore conducted this systematic review to synthesize the evidence on fine-tuning, RAG, and hybrid posttraining methods for clinical language models. Specifically, our objectives included (1) for which clinical task types does each adaptation strategy produce the largest performance gains relative to a nonadapted model or other comparator; (2) which design choices, including corpus scope, chunking, embedding model, prompting structure, and parameter scale, moderate those gains; and (3) how robust is the underlying evidence when judged by external validation, safety and fairness reporting, and risk of bias (ROB).</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>This systematic review was conducted in accordance with the PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) 2020 guidelines [<xref ref-type="bibr" rid="ref22">22</xref>] to synthesize evidence regarding posttraining and retrieval-augmented strategies designed to improve the performance, reliability, and clinical applicability of LLMs for clinical decision-making and outcomes. The protocol was submitted to PROSPERO (International Prospective Register of Systematic Reviews) in February 2026, and the registration (CRD420261308522) was finalized in May 2026, with screening, data extraction, and manuscript preparation proceeding from initial submission. A systematic framework was selected because the review addresses which categories of clinical task posttraining adaptation improve language model performance relative to nonadapted models, clinicians, or alternative decision-support systems.</p></sec><sec id="s2-2"><title>Search Strategy and Information Sources</title><p>We conducted a systematic search in PubMed or MEDLINE, Scopus, and Web of Science from January 2018 to May 2026. The year 2018 was selected as the inception date because it corresponds with the emergence of the transformer era in natural language processing, which enabled advances in generative AI for health care applications. The search strategies were combined MeSH and EMTREE terms and free-text words spanning across three concept domains that include (1) language model architectures and generative AI systems including &#x201C;large language model,&#x201D; &#x201C;transformer,&#x201D; &#x201C;GPT,&#x201D; &#x201C;LLaMA,&#x201D; &#x201C;Mistral,&#x201D; &#x201C;Claude,&#x201D; &#x201C;Qwen,&#x201D; &#x201C;Gemini,&#x201D; &#x201C;Gemma&#x201D; &#x201C;generative AI&#x201D;; (2) posttraining and adaptation techniques, including &#x201C;fine-tuning,&#x201D; &#x201C;instruction tuning,&#x201D; &#x201C;retrieval-augmented generation,&#x201D; &#x201C;RAG,&#x201D; &#x201C;Supervised Fine-Tuning,&#x201D; &#x201C;SFT,&#x201D; &#x201C;Direct Preference Optimization,&#x201D; &#x201C;Direct Preference Optimization,&#x201D; &#x201C;DPO,&#x201D; &#x201C;Low-Rank Adaptation,&#x201D; &#x201C;LoRA,&#x201D; &#x201C;parameter efficient fine tuning,&#x201D; &#x201C;PEFT&#x201D;; and (3) clinical applications and outcomes, including &#x201C;clinical decision support,&#x201D; &#x201C;diagnostic accuracy,&#x201D; &#x201C;differential diagnosis,&#x201D; &#x201C;triage,&#x201D; &#x201C;risk stratification,&#x201D; &#x201C;medical reasoning.&#x201D; We used Boolean operators, truncation syntax, and proximity operators to index each database. There were no language or publication-type filters other than those specified in the eligibility criteria during the search stage. The complete search strategy for each database is provided in Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-3"><title>Eligibility Criteria</title><p>Studies were included if they met all the following criteria: (1) studies were required to involve clinical datasets derived from electronic health records, radiology or pathology reports, laboratory results, discharge summaries, structured symptom descriptions, clinical vignettes, clinical textbooks, or multimodal clinical datasets across any health care domain; and (2) the study was required to evaluate at least one transformer-based language model, including but not limited to GPT, LLaMA (Meta AI), Mistral, Claude (Anthropic), Gemini (Google DeepMind), or related architectures that had undergone a posttraining adaptation or retrieval augmentation strategy. Eligible adaptation approaches included supervised fine-tuning (SFT), instruction tuning, domain adaptation, reinforcement learning or alignment optimization, direct preference optimization, parameter-efficient fine-tuning (PEFT), low-rank adaptation (LoRA), quantized low-rank adaptation (QLoRA), RAG, multimodal retrieval pipelines, or hybrid architectures integrating retrieval and fine-tuning mechanisms. Studies using retrieval augmentation without parameter updating were also eligible when retrieval mechanisms functioned as model-enhancement strategies intended to improve factual grounding, guideline concordance, retrieval fidelity, or hallucination mitigation in medical decision-support tasks such as diagnosis, differential diagnosis generation, clinical reasoning, triage, risk stratification, treatment recommendation support, or prognosis prediction. We also included studies addressing preoperative and postoperative clinical decision-making in surgical or perioperative contexts, including diagnosis, complication detection, risk stratification, and treatment-planning support. Studies were excluded when the model&#x2019;s target was intraoperative execution, such as instrument guidance, operative-step recognition, or technical performance assessment. We use the term language model throughout for transformer-based generative and encoder architectures, encompassing both LLMs and smaller domain-specific language models. Preprints were eligible if they reported complete methods and quantitative results, and were assessed for ROB on the same basis.</p><p>Studies were excluded if they evaluated a baseline language model without any posttraining adaptation; or (1) used traditional machine learning systems or deep-learning systems without a language model component; or (2) used chatbot or theoretical frameworks without implementation and evaluation; or (3) evaluated prompt engineering in isolation without performance data or reported only nonclinical computational benchmarks; or (4) were conducted in surgical or perioperative settings but focused primarily on the technical conduct or intraoperative performance of a surgical procedure rather than on diagnostic or decision-support tasks; or (5) were reviews, editorials, commentaries, conference abstracts, or study protocols without results were excluded.</p><p>The studies were required to report at least one evaluative performance outcome from the following categories: diagnostic accuracy (top-1 or top-k), sensitivity and/or specificity, area under the receiver operating characteristic curve (AUROC) or area under the precision-recall curve, differential diagnosis ranking accuracy, triage or risk stratification accuracy, scored clinical reasoning quality, calibration metrics, hallucination or erroneous recommendation rates, time-to-diagnosis, or results from external validation cohorts. The comparators included non-posttrained language models, human clinicians at any level of training, other clinical decision-support systems, alternative AI or machine learning&#x2013;diagnostic systems, and standard-of-care diagnosis.</p></sec><sec id="s2-4"><title>Screening and Selection</title><p>All identified records were imported into the Covidence (Veritas Health Innovation Ltd) systematic review software to facilitate collaborative screening and reviewer blinding. Title and abstract screening were performed independently by 2 reviewers (AP and YK) after automated deduplication, using predefined eligibility criteria. Discrepancies in selection were resolved through structured discussion or adjudication by a third reviewer (SKV).</p></sec><sec id="s2-5"><title>Data Extraction</title><p>The data extraction was performed using a standardized extraction instrument within Covidence. The following data elements were extracted for each included study: publication year, country, study design, clinical setting, health care setting, patient population or benchmark context, model openness (open-source, open-weight, or proprietary), base-model architecture, parameter scale, modality (text-only or multimodal), enhancement strategy, prompting methodology, retrieval infrastructure, external knowledge source, embedding models, vector databases, benchmark type, dataset characteristics, validation design, external testing procedures, explainability mechanisms, hallucination mitigation approaches, alignment and safety strategies, computational infrastructure, and primary performance outcomes. Data were extracted independently by 2 reviewers (AP and YK), with discrepancies adjudicated by consensus. For the included study coauthored by members of the review team, screening, data extraction, and ROB assessment were performed independently by YK, with the coauthoring reviewers excluded from its assessment.</p></sec><sec id="s2-6"><title>Quality Assessment and Analysis</title><p>Given the nature of the included studies (primarily diagnostic accuracy and model comparison studies), the ROB was assessed using PROBAST+AI (Prediction model Risk of Bias Assessment Tool for AI [<xref ref-type="bibr" rid="ref23">23</xref>]). PROBAST+AI was selected because the included studies primarily evaluated posttrained models for diagnostic reasoning, clinical decision support, disease classification, triage, and risk stratification tasks. We conducted our ROB assessment across four predefined methodological domains: (1) participants and data sources, (2) predictor handling, (3) outcome definition, and (4) analytical methodology. These domains were rated as having low, high, or unclear ROB according to the PROBAST+AI framework, along with applicability concerns and the reasoning for any identified bias [<xref ref-type="bibr" rid="ref23">23</xref>]. The ROB is summarized in Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-7"><title>Synthesis Approach</title><p>There was substantial clinical, methodological, and statistical heterogeneity across the included studies due to the diversity in model architecture, posttraining strategies, clinical tasks, outcome metrics, and evaluation of cohort characteristics. Therefore, the studies were grouped according to primary posttraining strategy: (1) SFT or PEFT or instruction tuning, (2) RAG, and (3) hybrid or multicomponent strategies. A meta-analysis was inappropriate due to the heterogeneity of these technologies and clinical contexts. Data were, therefore, descriptively summarized and synthesized for study characteristics, enhancement strategies, evaluation methods, and clinical performance outcomes. We elaborated on the performance relative to base LLMs and comparators, clinical domain coverage, evidence of external validation, and safety signals within each subgroup. When studies reported text-similarity metrics alongside clinical-accuracy outcomes, these are reported separately throughout this review and on a consistently labeled scale; text-similarity scores reflect surface-level overlap with a reference text and should not be interpreted as measures of diagnostic or clinical correctness.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Study Selection</title><p>The search returned 1890 records. After deduplication and automated filtering, 998 records were screened by title and abstract, of which 925 were excluded as off topic. Full texts were sought for 73 reports; of these, 38 were excluded (28 wrong study design and 10 wrong outcome). The remaining 35 studies published between 2024 and 2026 met inclusion criteria (<xref ref-type="fig" rid="figure1">Figure 1</xref>) [<xref ref-type="bibr" rid="ref24">24</xref>-<xref ref-type="bibr" rid="ref58">58</xref>]. Key characteristics of included studies are summarized in Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) flow diagram of study identification, screening, eligibility assessment, and the inclusion process for the systematic review.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e104092_fig01.png"/></fig></sec><sec id="s3-2"><title>Study Characteristics</title><p>Across the included 35 studies, the evaluated enhancement strategies fell into 3 families: SFT or PEFT (n=7, 20%), RAG (n=17, 48.6%), and hybrid pipelines that combined RAG and fine-tuning with structured prompting (n=11, 31.4%; <xref ref-type="fig" rid="figure2">Figure 2</xref>) [<xref ref-type="bibr" rid="ref24">24</xref>-<xref ref-type="bibr" rid="ref58">58</xref>]. In-silico benchmarks predominated (n=22), followed by retrospective electronic health record&#x2013;based model-development studies (n=8), prospective proof-of-concept evaluations (n=4), and 1 multicenter retrospective cohort. Clinical domains were heterogeneous, with the most frequent being oncology (n=5), diagnostic radiology (n=3), and neurological and cognitive disorders (n=5); mental health, cardiology, dermatology, ophthalmology, and emergency triage each contributed 2 studies; surgical specialties (perioperative care, hand surgery, microsurgery, and dentistry or oral surgery) contributed 5; and single studies covered pediatrics, rare disease, rehabilitation, urology, sleep medicine, pharmacy, and endocrinology. Studies originated from 12 countries and 1 international federated consortium, with the largest contributions from the United States (n=9), China (n=7), Germany (n=6), and multinational collaborations (n=4); the remainder came from Japan and Singapore (n=2, each), and from Israel, Italy, Canada, Thailand, and Taiwan (n=1, each).</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Mapping the landscape of clinical language model adaptation strategies by publication year, enhancement strategy, model openness, clinical domain, and data modality. Ribbon width is proportional to the number of included studies. Strategy nodes report study counts and cohort-size statistics; the right panels summarize method and modality composition within each clinical domain. PEFT: parameter-efficient fine-tuning; RAG: retrieval-augmented generation; SFT: supervised fine-tuning.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e104092_fig02.png"/></fig><p>Thirteen studies used proprietary frontier models (GPT-3.5 to GPT-4.5, OpenAI o1 or o3-mini, Claude 3 and 3.5, Gemini 1.5/2.0, Med-PaLM 2, DeepSeek R1, Grok 3 [SpaceXAI]). Seventeen used open-source or open-weight models (LLaMA-2/3 at 7B-70B, Mistral and Mixtral 8&#x00D7;7B, Gemma 2 27B [Google], Qwen 2/2.5/3 [Alibaba Cloud] at 4B-235B, ChatGLM-6B [Zhipu AI] and GLM4-9B [Zhipu AI], Japanese BERT [bidirectional encoder representations from transformers], GatorTron, and InstructBLIP-FLAN-T5-XL [Salesforce AI Research]). Five used mixed or hybrid model stacks. Parameter sizes ranged from approximately 110M to 235B, with 7B to 14B configurations most common. Eight studies were multimodal, integrating text with magnetic resonance imaging, computed tomography, dermoscopy, panoramic radiographs, electrocardiogram (ECG) images, or PubMed figures and tables.</p></sec><sec id="s3-3"><title>Adaptation Strategies</title><p>Seven studies used fine-tuning as the primary strategy, with LoRA and its variant QLoRA emerging as the de facto standard for parameter-efficient adaptation [<xref ref-type="bibr" rid="ref24">24</xref>-<xref ref-type="bibr" rid="ref30">30</xref>]. Fine-tuning was most effective for narrow, well-defined classification tasks where labeled data could be assembled at scale. An instruction-tuned LLaMA-2-7B model classified circulating-tumor-DNA fragmentome features with an external AUROC of 0.912 for cancer detection and 0.938 for hepatocellular carcinoma [<xref ref-type="bibr" rid="ref24">24</xref>]. A fine-tuned Japanese BERT model flagged newly identified acute infarcts on free-text radiology reports with macrosensitivity of 0.918 and an inference time of 0.115 seconds per patient [<xref ref-type="bibr" rid="ref26">26</xref>]. At the lower data extreme, a vision fine-tune of GPT-4o trained on only 20 labeled ECGs achieved 79.9% accuracy for detecting reduced left ventricular ejection fraction surpassing average clinician performance, although a dedicated convolutional network remained superior at 89.1% [<xref ref-type="bibr" rid="ref25">25</xref>]. Fine-tuned GPT-3 generated pediatric differential diagnoses comparable to pediatricians (87.3% vs 91.3%; <italic>P</italic>=.47), using only 350 rural-clinic encounters [<xref ref-type="bibr" rid="ref29">29</xref>]. At the opposite end of the data spectrum, large-scale instruction tuning of LLaMA 3.1-70B on 274,348 UK Biobank participants outperformed conventional machine learning baselines for major depressive disorder, with an area under curve of 0.892 [<xref ref-type="bibr" rid="ref30">30</xref>].</p><p>RAG was the most common standalone strategy, used in 17 studies, and was applied to 2 dominant use cases: guideline-grounded question answering and literature-based decision support [<xref ref-type="bibr" rid="ref31">31</xref>-<xref ref-type="bibr" rid="ref47">47</xref>]. Gains over nonaugmented baselines were largest when the retrieval corpus was authoritative and tightly scoped to the clinical question. Embedding the 2020 European Society of Cardiology acute coronary syndrome guideline raised DeepSeek R1 accuracy from 78.9% to 94.7% and ChatGPT-4o from 71.1% to 92.1% [<xref ref-type="bibr" rid="ref34">34</xref>]. A trauma-radiology chatbot grounded in a curated reading list improved diagnostic accuracy from 93% to 100% and grading accuracy from 48% to 87% [<xref ref-type="bibr" rid="ref38">38</xref>]. A guideline-grounded urology pipeline reached 95.5% concordance for prostate-specific&#x2013;antigen testing recommendations, compared with 62.3% closed-book and 74.1% open-book accuracy among junior clinicians [<xref ref-type="bibr" rid="ref45">45</xref>]. Specialty-corpus RAG also produced consistent benefits in ophthalmology [<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref35">35</xref>], rare disease diagnosis [<xref ref-type="bibr" rid="ref33">33</xref>], microsurgery and hand surgery [<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref43">43</xref>], and radiation oncology [<xref ref-type="bibr" rid="ref44">44</xref>]. Notably, RAG did not always help. In a 2000-case MIMIC-IV (Medical Information Mart for Intensive Care-IV) evaluation, the strongest standalone Claude 3.5 Sonnet workflow outperformed its RAG-augmented counterpart on overall accuracy [<xref ref-type="bibr" rid="ref39">39</xref>], and a German emergency-department study found that semantic retrieval introduced formal errors in 23% of responses despite faster retrieval, leaving the nonretrieval Mixtral baseline highest in physician-rated usefulness [<xref ref-type="bibr" rid="ref40">40</xref>]. Two important moderators of RAG performance were model scale and reasoning capacity. Smaller models were destabilized by retrieval noise unless the corpus was structured and a hybrid sparse-dense index was used; structured table-of-contents&#x2013;aligned chunks combined with BM25 plus PubMedBERT-dense retrieval reversed a 7.1% accuracy drop in Llama-3-8B and produced a 6.1% mean Top-1 gain on clinical sleep-medicine cases [<xref ref-type="bibr" rid="ref47">47</xref>]. Reasoning models such as DeepSeek R1 and OpenAI o-series gained little or nothing from retrieval, suggesting that internal chain-of-thought capacity can substitute for some forms of external grounding [<xref ref-type="bibr" rid="ref44">44</xref>].</p></sec><sec id="s3-4"><title>Hybrid and Multicomponent Pipelines</title><p>Eleven studies combined fine-tuning, retrieval, and structured prompting into a single pipeline, and these hybrid systems consistently produced the highest clinical-utility scores for complex tasks that required both representation learning and external knowledge grounding. A ChatGLM-6B system that combined LoRA instruction tuning with tool chaining and structured emergency notes correctly classified stroke vs nonstroke in 99.0% of internal cases and 95.5% and 79.1% in 2 external cohorts, with ischemia-vs-hemorrhage accuracy above 97% even on external data [<xref ref-type="bibr" rid="ref53">53</xref>]. A multimodal Qwen2-VL pipeline for osteonecrosis of the jaw integrated panoramic-radiograph segmentation, LoRA fine-tuning, and visual question answering to reach 96.0% expert-rated accuracy, exceeding junior surgeons (88.4%) and approaching senior surgeons (99.6%) [<xref ref-type="bibr" rid="ref48">48</xref>]. A federated multimodal dermatology system achieved 90.2% diagnostic accuracy across 11 lesion types and 93.3% benign-vs-malignant accuracy on a 4452-image external Stanford and MIDAS (Multimodal Image Dataset for AI-based Skin Cancer) cohort, while keeping image processing local [<xref ref-type="bibr" rid="ref57">57</xref>]. A liver-cancer assistant that fused small-model image features, retrieval over the CSCO (Chinese Society of Clinical Oncology) Liver Cancer Guidelines, and a doctor-style 3-step chain-of-thought prompt raised expert-quality scores for image interpretation from 5.9 to 7.2 and treatment-plan reasonableness from 4.2 to 6.5 [<xref ref-type="bibr" rid="ref54">54</xref>]. Other hybrid systems demonstrated similar synergies in Alzheimer disease [<xref ref-type="bibr" rid="ref51">51</xref>,<xref ref-type="bibr" rid="ref52">52</xref>], ophthalmology [<xref ref-type="bibr" rid="ref50">50</xref>], brain-metastasis magnetic resonance imaging reporting [<xref ref-type="bibr" rid="ref55">55</xref>], laryngeal-cancer Bayesian network modeling [<xref ref-type="bibr" rid="ref56">56</xref>], and outpatient diabetes decision support [<xref ref-type="bibr" rid="ref58">58</xref>].</p></sec><sec id="s3-5"><title>Importance of Prompting Strategy</title><p>Prompting was treated in many studies as a first-class enhancement lever rather than a secondary detail, and prompt design changes alone produced clinically meaningful shifts in performance. Adding a brief expert-persona preprompt raised guideline-based dental endocarditis-prophylaxis accuracy from 83.6% to 90.0% across 7 frontier models, and progressive least-to-most prompting outperformed simple prompts for chronic low-back&#x2013;pain treatment recommendations [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref46">46</xref>]. For rare-disease diagnosis, switching from a base prompt to a prompt-with-explanation template increased GPT-3.5 accuracy from 40% to 43% within the same retrieval pipeline [<xref ref-type="bibr" rid="ref33">33</xref>]. Structured chain-of-thought prompts were used in nearly every hybrid system; in a perioperative-complication pipeline, the combination of a &#x201C;think&#x201D; reasoning field, structured JSON outputs, and targeted single-complication prompts more than doubled micro-<italic>F</italic><sub>1</sub> in a 4-billion-parameter Qwen model [<xref ref-type="bibr" rid="ref49">49</xref>]. Multistep, role-based prompting mirroring clinician reasoning was central to Wu-2025-liver-cancer, Tung-2025-PSA-RAG, and Lammert-2024-MTB-CoT, and an adaptive prompt-refinement workflow lifted brain-metastasis detection sensitivity from 0.84 to 0.98 [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref55">55</xref>]. At the same time, prompting had a clear ceiling: when the underlying model lacked the relevant clinical knowledge or the retrieval corpus was inadequate, prompting changes alone could not close the gap [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref40">40</xref>]. The practical pattern was that small, low-cost changes in persona, chain-of-thought, structured output schemas, and example formatting routinely shifted accuracy by 5 to 15 percentage points and frequently determined whether a downstream evaluation crossed clinically meaningful thresholds.</p></sec><sec id="s3-6"><title>Datasets and Knowledge Sources</title><p>Dataset quality, structure, and curation mattered as much as raw size. Training corpora ranged from 20 ECG images [<xref ref-type="bibr" rid="ref25">25</xref>] to 274,348 biobank participants [<xref ref-type="bibr" rid="ref30">30</xref>], and retrieval corpora ranged from 96 peer-reviewed articles [<xref ref-type="bibr" rid="ref32">32</xref>] to 30 million PubMed abstracts [<xref ref-type="bibr" rid="ref39">39</xref>]. Across studies, 3 patterns were consistent. First, well-curated, narrowly-scoped corpora outperformed broader corpora for guideline-bound questions; the Tung-2025 prostate-specific antigen pipeline relied on a 239-page guideline set, and the Alexandrou-2025 acute coronary syndrome pipeline relied on a single guideline document, yet both produced near-ceiling accuracies [<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref45">45</xref>]. Second, chunking strategy and embedding model choice were major performance levers. A laryngeal-cancer system raised retrieval accuracy from 0.75 to 0.90 by combining recursive chunking with a fine-tuned general text embeddings&#x2013;large embedding model [<xref ref-type="bibr" rid="ref56">56</xref>], and table-of-contents&#x2013;aligned 512-token segments outperformed raw text segmentation in sleep medicine [<xref ref-type="bibr" rid="ref47">47</xref>]. Domain-tuned or biomedical embeddings (PubMedBERT-dense [Microsoft Research], bge-small-en [Beijing Academy of Artificial Intelligence], fine-tuned general text embeddings&#x2013;large) generally outperformed off-the-shelf OpenAI embeddings in retrieval-quality metrics. Third, external validation often revealed substantial performance drops, indicating site-specific adaptation may still be necessary even after fine-tuning or RAG; the Song-2025 stroke model fell from 99.0% internal accuracy to 79.1% in 1 external cohort. Privacy-preserving designs, including federated learning [<xref ref-type="bibr" rid="ref57">57</xref>], local on-premises deployment of small, fine-tuned models [<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref58">58</xref>], and institution-owned retrieval indexes [<xref ref-type="bibr" rid="ref41">41</xref>] were a notable design trend in 2025 to 2026 studies, suggesting growing attention to deployment in regulated clinical environments.</p></sec><sec id="s3-7"><title>External Validation, Safety, and ROB</title><p>PROBAST+AI assessment showed that 25 of 35 (71.4%) studies were judged to be at high ROB, primarily owing to limited external validation, inadequate calibration assessment, and incomplete methodological reporting. Nine (25.7%) studies had an unclear ROB because key analytical details were insufficiently reported, while only 1 (2.9%) study was rated as low ROB across all domains. Most relied exclusively on internal validation using held-out datasets, retrospective cohorts, benchmark datasets, expert-scored vignettes, or synthetic cases and only a minority of studies performed external validation. These findings highlight substantial methodological limitations in the current evidence base and underscore the need for more rigorous validation and transparent reporting of clinical language model adaptation studies. Full domain-level ratings are presented in Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>In this systematic review of 35 studies, posttraining and retrieval-based adaptation consistently improved the clinical performance of language models across a broad range of tasks. However, the magnitude and reliability of improvement depended heavily on the clinical task, data source, prompting strategy, retrieval design, and system architecture. Several main findings emerged. First, SFT and PEFT were most effective for narrow, well-defined classification or generation tasks supported by labeled clinical data. Second, RAG was most effective for guideline-bound or literature-intensive questions, particularly when the retrieval corpus was accurate, relevant, and well-organized. Third, hybrid systems that combined fine-tuning, retrieval, structured prompting, and, in some cases, multimodal feature extraction appeared most suitable for complex clinical decision-support tasks requiring both learned clinical pattern recognition and access to external medical knowledge. Fourth, prompting was not merely a technical detail but an important performance lever, with structured prompts, expert-role framing, chain-of-thought reasoning, and task decomposition often improving model consistency and clinical usefulness. Fifth, dataset quality, corpus structure, chunking strategy, embedding choice, and external validation were central determinants of performance, often mattering as much as model size. These findings suggest that clinical language model adaptation should not be viewed as a competition between these strategies. Rather, each approach addresses a different limitation of general-purpose models.</p><p>Fine-tuning improves task-specific behavior by exposing the model to labeled examples from a defined clinical distribution [<xref ref-type="bibr" rid="ref59">59</xref>]. This was most evident in studies focused on cancer detection from cfDNA-derived features, acute infarct identification from radiology reports, pediatric differential diagnosis generation, cognitive decline detection, and major depressive disorder classification [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref30">30</xref>]. In these settings, the model was asked to perform a constrained task with a relatively clear reference standard. Even modest datasets were often sufficient to improve performance, especially when the task was narrow and the input format was standardized. This supports the practical value of parameter-efficient methods such as LoRA, which can adapt smaller open-weight models without the cost, privacy burden, or infrastructure needs of full model retraining. Across studies, LoRA generally performed comparably to or slightly better than QLoRA, and several reports suggested that even a few hundred well-curated examples were often sufficient to specialize a foundation model for a single clinical task.</p><p>RAG addressed the need to ground model output in external, updated, and domain-specific knowledge. The largest gains were seen when the clinical question mapped closely to a trusted corpus [<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref47">47</xref>]. In these cases, RAG improved not only answer accuracy but also the traceability of recommendations. This is clinically important because many errors from general models arise not from language fluency but from outdated, incomplete, or nonspecific medical knowledge. However, RAG was not uniformly beneficial. Some studies showed that poorly matched retrieval, noisy chunks, or overly broad corpora could distract the model and worsen performance [<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref47">47</xref>]. Thus, retrieval should be treated as a clinical engineering problem rather than a simple add-on. Corpus selection, chunking strategy, embedding model choice, reranking, and postretrieval filtering may determine whether RAG improves or degrades clinical output. Performance was also influenced by model scale and reasoning capacity, with smaller models being more sensitive to retrieval noise unless structured chunking and hybrid sparse&#x2013;dense indexing were used. In contrast, models with strong reasoning showed limited additional benefit from retrieval, suggesting that strong internal reasoning may partially substitute for external grounding in some settings.</p><p>Hybrid systems appeared most promising for high-complexity tasks. Real-world decision support rarely depends on a single capability. It requires recognition of patient-specific patterns, knowledge of guidelines, interpretation of multimodal data, and generation of usable recommendations. Hybrid systems may therefore represent the most realistic architecture for deployment, particularly when the goal extends beyond question answering to workflow-level clinical decision support. Across studies, no single component consistently dominated; instead, performance depended on how well fine-tuning, retrieval, prompting, and multimodal inputs were jointly aligned with the clinical task. Three cross-cutting practical principles emerged among different methods. PEFT, particularly LoRA, was the dominant fine-tuning approach and enabled privacy-preserving on-premises deployment of small open-weight models. Embedding choice, chunking, and prompt engineering were comparatively low-cost levers that frequently mattered more than choosing a larger base model. Reasoning-class models reduced but did not eliminate the marginal benefit of retrieval, suggesting that the optimal architecture is increasingly dependent on the specific clinical task rather than on any single dominant adaptation strategy.</p><p>A second important observation is that smaller, locally deployable models can perform well when adaptation is carefully designed. Several studies showed that compact open-weight models could approach or exceed larger proprietary systems after targeted fine-tuning, task decomposition, or domain-specific retrieval [<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref58">58</xref>]. This has practical implications for health care systems, where data privacy, latency, cost, auditability, and local governance are central barriers to implementation. Open-weight models adapted within institutional environments may offer a more feasible pathway for many clinical applications than reliance on externally hosted general-purpose models. The emergence of federated and privacy-preserving approaches further suggests that multi-institutional model development may be possible without centralizing sensitive patient data [<xref ref-type="bibr" rid="ref57">57</xref>].</p><p>Prompting also emerged as a meaningful determinant of performance. Persona prompts, structured chain-of-thought-style reasoning, JSON output schemas, least-to-most prompting, and task decomposition produced measurable gains across several studies [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref55">55</xref>]. These findings should not be interpreted to mean that prompting alone is sufficient for clinical reliability. Instead, prompt design appears to function as an interface between the clinical task and the model. It can improve consistency, enforce structure, and reduce ambiguity, but it cannot compensate for poor retrieval, weak reference standards, or lack of domain knowledge. For clinical deployment, prompts should be version-controlled, tested across cases, and reported with enough detail to allow reproducibility.</p><p>Despite encouraging results, the current evidence base remains early. Most included studies were retrospective, in-silico, or proof-of-concept evaluations. Only a small number used prospective designs, and even fewer assessed real-world workflow integration, clinician behavior, patient outcomes, or downstream safety. Many studies reported accuracy, <italic>F</italic><sub>1</sub>-score, AUROC, or expert-rated quality, but fewer evaluated calibration, uncertainty, subgroup performance, hallucination rates, or harmful recommendations. External validation was inconsistent, and, when performed, performance sometimes dropped substantially across sites or cohorts. These findings underscore that benchmark performance alone is insufficient for clinical readiness. For decision support systems, future studies should assess not only whether the model is correct but also when it is wrong, whether it knows when to abstain, how errors affect clinicians, and whether use of the system improves patient-relevant outcomes.</p></sec><sec id="s4-2"><title>Limitations</title><p>This review has limitations. First, meta-analysis was not feasible because of substantial heterogeneity in clinical domains, model architectures, adaptation methods, outcome metrics, and evaluation designs. Our study reports the direction and consistency of reported effects rather than pooled estimates, and the descriptive characterization of strategies, clinical domains, and model families shares features with evidence mapping. Second, most studies were model-development or benchmark studies rather than prospective clinical trials, limiting inference about real-world effectiveness. Third, many studies used simulated vignettes, synthetic cases, or curated datasets, which may overestimate performance compared with routine clinical environments. Fourth, reporting of calibration, fairness, demographic subgroup performance, and safety outcomes was inconsistent. Fifth, the rapid pace of LLM development means that model versions, retrieval tools, and fine-tuning methods may change quickly, limiting the durability of model-specific conclusions. Sixth, although the search combined generic architecture terms with the named-model terms, it was necessarily incomplete, as relevant studies are not consistently described as LLMs in titles or abstracts. We did not assess reporting bias formally, as no established method applies to this class of study, and these findings should be read as indicating the direction rather than the magnitude of benefit. However, the broader task-to-method patterns identified in this review are likely to remain relevant as clinical LLM systems continue to evolve.</p></sec><sec id="s4-3"><title>Conclusions</title><p>In clinical practice, the performance of LLMs depends less on any single technical approach and more on how these tools are combined to support real clinical decision-making. Fine-tuning helps models adapt to specific clinical tasks, retrieval grounds outputs in current medical knowledge, structured prompting improves clarity and consistency of responses, and multimodal inputs allow incorporation of imaging and other patient data. When used together appropriately, these approaches can improve the usefulness of AI systems in supporting clinicians, but their value ultimately depends on how well they fit into clinical workflows and decision needs. Future research should move beyond isolated benchmark improvements toward prospective, externally validated, and clinically embedded evaluations that prioritize safety, reliability, and patient-centered outcomes.</p></sec></sec></body><back><ack><p>The preparation of this manuscript did not involve the use of any generative AI services.</p></ack><notes><sec><title>Funding</title><p>This work was partially supported by the National Institutes of Health (NIH) under award numbers R01AG084236 and U01AG088076.</p></sec><sec><title>Data Availability</title><p>Summary characteristics, complete database search strategies, and risk-of-bias assessments are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: AP, YK, JC</p><p>Data curation: AP, YK</p><p>Formal analysis: AP, SKV, PL</p><p>Investigation: AP, YK, SKV</p><p>Methodology: AP, YK, JC</p><p>Software: AP, SKV, YK</p><p>Supervision: CT, JC</p><p>Validation: AP, YK, PL, CT, JC</p><p>Visualization: AP, YK, SKV</p><p>Writing &#x2013; original draft: AP, YK, SKV</p><p>Writing &#x2013; review &#x0026; editing: AP, YK, SKV, PL, CT, JC</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AUROC</term><def><p>area under the receiver operating characteristic curve</p></def></def-item><def-item><term id="abb2">BERT</term><def><p>bidirectional encoder representations from transformers</p></def></def-item><def-item><term id="abb3">CSCO</term><def><p>Chinese Society of Clinical Oncology</p></def></def-item><def-item><term id="abb4">ECG</term><def><p>electrocardiogram</p></def></def-item><def-item><term id="abb5">IDEA</term><def><p>Interpretive Summary, Differential Diagnosis, Explanation of Reasoning, and Alternatives</p></def></def-item><def-item><term id="abb6">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb7">LoRA</term><def><p>low-rank adaptation</p></def></def-item><def-item><term id="abb8">MIDAS</term><def><p>Multimodal Image Dataset for AI-based Skin Cancer</p></def></def-item><def-item><term id="abb9">MIMIC-IV</term><def><p>Medical Information Mart for Intensive Care-IV</p></def></def-item><def-item><term id="abb10">PEFT</term><def><p>parameter-efficient fine-tuning</p></def></def-item><def-item><term id="abb11">PRISMA</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses</p></def></def-item><def-item><term id="abb12">PROBAST+AI</term><def><p>Prediction model Risk of Bias Assessment Tool for AI</p></def></def-item><def-item><term id="abb13">PROSPERO</term><def><p>International Prospective Register of Systematic Reviews</p></def></def-item><def-item><term id="abb14">QLoRA</term><def><p>quantized low-rank adaptation</p></def></def-item><def-item><term id="abb15">RAG</term><def><p>retrieval-augmented generation</p></def></def-item><def-item><term id="abb16">ROB</term><def><p>risk of bias</p></def></def-item><def-item><term id="abb17">SFT</term><def><p>supervised fine-tuning</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rajkomar</surname><given-names>A</given-names> </name><name name-style="western"><surname>Dean</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kohane</surname><given-names>I</given-names> </name></person-group><article-title>Machine learning in medicine</article-title><source>N Engl J Med</source><year>2019</year><month>04</month><day>4</day><volume>380</volume><issue>14</issue><fpage>1347</fpage><lpage>1358</lpage><pub-id pub-id-type="doi">10.1056/NEJMra1814259</pub-id><pub-id pub-id-type="medline">30943338</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Esteva</surname><given-names>A</given-names> </name><name name-style="western"><surname>Robicquet</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ramsundar</surname><given-names>B</given-names> </name><etal/></person-group><article-title>A guide to deep learning in healthcare</article-title><source>Nat Med</source><year>2019</year><month>01</month><volume>25</volume><issue>1</issue><fpage>24</fpage><lpage>29</lpage><pub-id pub-id-type="doi">10.1038/s41591-018-0316-z</pub-id><pub-id pub-id-type="medline">30617335</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Topol</surname><given-names>EJ</given-names> </name></person-group><article-title>High-performance medicine: the convergence of human and artificial intelligence</article-title><source>Nat Med</source><year>2019</year><month>01</month><volume>25</volume><issue>1</issue><fpage>44</fpage><lpage>56</lpage><pub-id pub-id-type="doi">10.1038/s41591-018-0300-7</pub-id><pub-id pub-id-type="medline">30617339</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moor</surname><given-names>M</given-names> </name><name name-style="western"><surname>Banerjee</surname><given-names>O</given-names> </name><name name-style="western"><surname>Abad</surname><given-names>ZSH</given-names> </name><etal/></person-group><article-title>Foundation models for generalist medical artificial intelligence</article-title><source>Nature</source><year>2023</year><month>04</month><volume>616</volume><issue>7956</issue><fpage>259</fpage><lpage>265</lpage><pub-id pub-id-type="doi">10.1038/s41586-023-05881-4</pub-id><pub-id pub-id-type="medline">37045921</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shah</surname><given-names>NH</given-names> </name><name name-style="western"><surname>Entwistle</surname><given-names>D</given-names> </name><name name-style="western"><surname>Pfeffer</surname><given-names>MA</given-names> </name></person-group><article-title>Creation and adoption of large language models in medicine</article-title><source>JAMA</source><year>2023</year><month>09</month><day>5</day><volume>330</volume><issue>9</issue><fpage>866</fpage><lpage>869</lpage><pub-id pub-id-type="doi">10.1001/jama.2023.14217</pub-id><pub-id pub-id-type="medline">37548965</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thirunavukarasu</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSJ</given-names> </name><name name-style="western"><surname>Elangovan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Gutierrez</surname><given-names>L</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>TF</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSW</given-names> </name></person-group><article-title>Large language models in medicine</article-title><source>Nat Med</source><year>2023</year><month>08</month><volume>29</volume><issue>8</issue><fpage>1930</fpage><lpage>1940</lpage><pub-id pub-id-type="doi">10.1038/s41591-023-02448-8</pub-id><pub-id pub-id-type="medline">37460753</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Azizi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Large language models encode clinical knowledge</article-title><source>Nature</source><year>2023</year><month>08</month><volume>620</volume><issue>7972</issue><fpage>172</fpage><lpage>180</lpage><pub-id pub-id-type="doi">10.1038/s41586-023-06291-2</pub-id><pub-id pub-id-type="medline">37438534</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ayers</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Poliak</surname><given-names>A</given-names> </name><name name-style="western"><surname>Dredze</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Comparing physician and artificial intelligence chatbot responses to patient questions posted to a public social media forum</article-title><source>JAMA Intern Med</source><year>2023</year><month>06</month><day>1</day><volume>183</volume><issue>6</issue><fpage>589</fpage><lpage>596</lpage><pub-id pub-id-type="doi">10.1001/jamainternmed.2023.1838</pub-id><pub-id pub-id-type="medline">37115527</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Van Veen</surname><given-names>D</given-names> </name><name name-style="western"><surname>Van Uden</surname><given-names>C</given-names> </name><name name-style="western"><surname>Blankemeier</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Adapted large language models can outperform medical experts in clinical text summarization</article-title><source>Nat Med</source><year>2024</year><month>04</month><volume>30</volume><issue>4</issue><fpage>1134</fpage><lpage>1142</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-02855-5</pub-id><pub-id pub-id-type="medline">38413730</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Gottweis</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Toward expert-level medical question answering with large language models</article-title><source>Nat Med</source><year>2025</year><month>03</month><volume>31</volume><issue>3</issue><fpage>943</fpage><lpage>950</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03423-7</pub-id><pub-id pub-id-type="medline">39779926</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>SF</given-names> </name><name name-style="western"><surname>Alyakin</surname><given-names>A</given-names> </name><name name-style="western"><surname>Seas</surname><given-names>A</given-names> </name><etal/></person-group><article-title>LLM-assisted systematic review of large language models in clinical medicine</article-title><source>Nat Med</source><year>2026</year><month>03</month><volume>32</volume><issue>3</issue><fpage>1152</fpage><lpage>1159</lpage><pub-id pub-id-type="doi">10.1038/s41591-026-04229-5</pub-id><pub-id pub-id-type="medline">41776077</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Agrawal</surname><given-names>M</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>IY</given-names> </name><name name-style="western"><surname>Gulamali</surname><given-names>F</given-names> </name><name name-style="western"><surname>Joshi</surname><given-names>S</given-names> </name></person-group><article-title>The evaluation illusion of large language models in medicine</article-title><source>NPJ Digit Med</source><year>2025</year><month>10</month><day>7</day><volume>8</volume><issue>1</issue><fpage>600</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01963-x</pub-id><pub-id pub-id-type="medline">41057566</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hager</surname><given-names>P</given-names> </name><name name-style="western"><surname>Jungmann</surname><given-names>F</given-names> </name><name name-style="western"><surname>Holland</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Evaluation and mitigation of the limitations of large language models in clinical decision-making</article-title><source>Nat Med</source><year>2024</year><month>09</month><volume>30</volume><issue>9</issue><fpage>2613</fpage><lpage>2622</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03097-1</pub-id><pub-id pub-id-type="medline">38965432</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kanjee</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Crowe</surname><given-names>B</given-names> </name><name name-style="western"><surname>Rodman</surname><given-names>A</given-names> </name></person-group><article-title>Accuracy of a generative artificial intelligence model in a complex diagnostic challenge</article-title><source>JAMA</source><year>2023</year><month>07</month><day>3</day><volume>330</volume><issue>1</issue><fpage>78</fpage><lpage>80</lpage><pub-id pub-id-type="doi">10.1001/jama.2023.8288</pub-id><pub-id pub-id-type="medline">37318797</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cabral</surname><given-names>S</given-names> </name><name name-style="western"><surname>Restrepo</surname><given-names>D</given-names> </name><name name-style="western"><surname>Kanjee</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Clinical reasoning of a generative artificial intelligence model compared with physicians</article-title><source>JAMA Intern Med</source><year>2024</year><month>05</month><day>1</day><volume>184</volume><issue>5</issue><fpage>581</fpage><lpage>583</lpage><pub-id pub-id-type="doi">10.1001/jamainternmed.2024.0295</pub-id><pub-id pub-id-type="medline">38557971</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Goh</surname><given-names>E</given-names> </name><name name-style="western"><surname>Gallo</surname><given-names>R</given-names> </name><name name-style="western"><surname>Hom</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Large language model influence on diagnostic reasoning: a randomized clinical trial</article-title><source>JAMA Netw Open</source><year>2024</year><month>10</month><day>1</day><volume>7</volume><issue>10</issue><fpage>e2440969</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.40969</pub-id><pub-id pub-id-type="medline">39466245</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Omar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sorin</surname><given-names>V</given-names> </name><name name-style="western"><surname>Collins</surname><given-names>JD</given-names> </name><etal/></person-group><article-title>Multi-model assurance analysis showing large language models are highly vulnerable to adversarial hallucination attacks during clinical decision support</article-title><source>Commun Med (Lond)</source><year>2025</year><month>08</month><day>2</day><volume>5</volume><issue>1</issue><fpage>330</fpage><pub-id pub-id-type="doi">10.1038/s43856-025-01021-3</pub-id><pub-id pub-id-type="medline">40753316</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Omiye</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Lester</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Spichak</surname><given-names>S</given-names> </name><name name-style="western"><surname>Rotemberg</surname><given-names>V</given-names> </name><name name-style="western"><surname>Daneshjou</surname><given-names>R</given-names> </name></person-group><article-title>Large language models propagate race-based medicine</article-title><source>NPJ Digit Med</source><year>2023</year><month>10</month><day>20</day><volume>6</volume><issue>1</issue><fpage>195</fpage><pub-id pub-id-type="doi">10.1038/s41746-023-00939-z</pub-id><pub-id pub-id-type="medline">37864012</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>G</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Closing the gap between open source and commercial large language models for medical evidence summarization</article-title><source>NPJ Digit Med</source><year>2024</year><month>09</month><day>9</day><volume>7</volume><issue>1</issue><fpage>239</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01239-w</pub-id><pub-id pub-id-type="medline">39251804</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>R</given-names> </name><name name-style="western"><surname>Ning</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Keppo</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Retrieval-augmented generation for generative artificial intelligence in health care</article-title><source>Npj Health Syst</source><year>2025</year><month>01</month><day>25</day><volume>2</volume><issue>1</issue><fpage>2</fpage><pub-id pub-id-type="doi">10.1038/s44401-024-00004-1</pub-id><pub-id pub-id-type="medline">42527447</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wada</surname><given-names>A</given-names> </name><name name-style="western"><surname>Tanaka</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Nishizawa</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Retrieval-augmented generation elevates local LLM quality in radiology contrast media consultation</article-title><source>NPJ Digit Med</source><year>2025</year><month>07</month><day>2</day><volume>8</volume><issue>1</issue><fpage>395</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01802-z</pub-id><pub-id pub-id-type="medline">40604147</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Page</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>McKenzie</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Bossuyt</surname><given-names>PM</given-names> </name><etal/></person-group><article-title>The PRISMA 2020 statement: an updated guideline for reporting systematic reviews</article-title><source>BMJ</source><year>2021</year><month>03</month><day>29</day><volume>372</volume><fpage>n71</fpage><pub-id pub-id-type="doi">10.1136/bmj.n71</pub-id><pub-id pub-id-type="medline">33782057</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moons</surname><given-names>KGM</given-names> </name><name name-style="western"><surname>Damen</surname><given-names>JAA</given-names> </name><name name-style="western"><surname>Kaul</surname><given-names>T</given-names> </name><etal/></person-group><article-title>PROBAST+AI: an updated quality, risk of bias, and applicability assessment tool for prediction models using regression or artificial intelligence methods</article-title><source>BMJ</source><year>2025</year><month>03</month><day>24</day><volume>388</volume><fpage>e082505</fpage><pub-id pub-id-type="doi">10.1136/bmj-2024-082505</pub-id><pub-id pub-id-type="medline">40127903</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>H</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>K</given-names> </name><name name-style="western"><surname>Li</surname><given-names>X</given-names> </name></person-group><article-title>Large language model produces high accurate diagnosis of cancer from end-motif profiles of cell-free DNA</article-title><source>Brief Bioinform</source><year>2024</year><month>07</month><day>25</day><volume>25</volume><issue>5</issue><fpage>bbae430</fpage><pub-id pub-id-type="doi">10.1093/bib/bbae430</pub-id><pub-id pub-id-type="medline">39222060</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Engelstein</surname><given-names>H</given-names> </name><name name-style="western"><surname>Ramon-Gonen</surname><given-names>R</given-names> </name><name name-style="western"><surname>Barbash</surname><given-names>I</given-names> </name><name name-style="western"><surname>Beinart</surname><given-names>R</given-names> </name><name name-style="western"><surname>Cohen-Shelly</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sabbag</surname><given-names>A</given-names> </name></person-group><article-title>Estimating LVEF from ECG with GPT-4o fine-tuned vision: a novel approach in AI-driven cardiac diagnostics</article-title><source>J Med Syst</source><year>2025</year><month>11</month><day>10</day><volume>49</volume><issue>1</issue><fpage>157</fpage><pub-id pub-id-type="doi">10.1007/s10916-025-02289-7</pub-id><pub-id pub-id-type="medline">41212334</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fujita</surname><given-names>N</given-names> </name><name name-style="western"><surname>Yasaka</surname><given-names>K</given-names> </name><name name-style="western"><surname>Kiryu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Abe</surname><given-names>O</given-names> </name></person-group><article-title>Fine-tuned large language model for extracting newly identified acute brain infarcts based on computed tomography or magnetic resonance imaging reports</article-title><source>Emerg Radiol</source><year>2025</year><month>08</month><volume>32</volume><issue>4</issue><fpage>495</fpage><lpage>501</lpage><pub-id pub-id-type="doi">10.1007/s10140-025-02354-1</pub-id><pub-id pub-id-type="medline">40451964</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guan</surname><given-names>H</given-names> </name><name name-style="western"><surname>Novoa-Laurentiev</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>L</given-names> </name></person-group><article-title>CD-Tron: leveraging large clinical language model for early detection of cognitive decline from electronic health records</article-title><source>J Biomed Inform</source><year>2025</year><month>06</month><volume>166</volume><fpage>104830</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2025.104830</pub-id><pub-id pub-id-type="medline">40320101</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Iinuma</surname><given-names>K</given-names> </name><name name-style="western"><surname>Fujii</surname><given-names>K</given-names> </name><name name-style="western"><surname>Nakashima</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Multiclass classification of pigmented skin lesions using a multimodal large language model</article-title><source>Cureus</source><year>2025</year><month>07</month><volume>17</volume><issue>7</issue><fpage>e88711</fpage><pub-id pub-id-type="doi">10.7759/cureus.88711</pub-id><pub-id pub-id-type="medline">40861687</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mansoor</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ibrahim</surname><given-names>AF</given-names> </name><name name-style="western"><surname>Grindem</surname><given-names>D</given-names> </name><name name-style="western"><surname>Baig</surname><given-names>A</given-names> </name></person-group><article-title>Large language models for pediatric differential diagnoses in rural health care: multicenter retrospective cohort study comparing GPT-3 with pediatrician performance</article-title><source>JMIRx Med</source><year>2025</year><month>03</month><day>19</day><volume>6</volume><fpage>e65263</fpage><pub-id pub-id-type="doi">10.2196/65263</pub-id><pub-id pub-id-type="medline">40106452</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sha</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Pan</surname><given-names>H</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>W</given-names> </name><etal/></person-group><article-title>MDD-LLM: towards accuracy large language models for major depressive disorder diagnosis</article-title><source>J Affect Disord</source><year>2025</year><month>11</month><day>1</day><volume>388</volume><fpage>119774</fpage><pub-id pub-id-type="doi">10.1016/j.jad.2025.119774</pub-id><pub-id pub-id-type="medline">40581100</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lammert</surname><given-names>J</given-names> </name><name name-style="western"><surname>Dreyer</surname><given-names>T</given-names> </name><name name-style="western"><surname>Mathes</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Expert-guided large language models for clinical decision support in precision oncology</article-title><source>JCO Precis Oncol</source><year>2024</year><month>10</month><volume>8</volume><fpage>e2400478</fpage><pub-id pub-id-type="doi">10.1200/PO-24-00478</pub-id><pub-id pub-id-type="medline">39475661</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rau</surname><given-names>S</given-names> </name><name name-style="western"><surname>Rau</surname><given-names>A</given-names> </name><name name-style="western"><surname>Nattenm&#x00FC;ller</surname><given-names>J</given-names> </name><etal/></person-group><article-title>A retrieval-augmented chatbot based on GPT-4 provides appropriate differential diagnosis in gastrointestinal radiology: a proof of concept study</article-title><source>Eur Radiol Exp</source><year>2024</year><month>05</month><day>17</day><volume>8</volume><issue>1</issue><fpage>60</fpage><pub-id pub-id-type="doi">10.1186/s41747-024-00457-x</pub-id><pub-id pub-id-type="medline">38755410</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zelin</surname><given-names>C</given-names> </name><name name-style="western"><surname>Chung</surname><given-names>WK</given-names> </name><name name-style="western"><surname>Jeanne</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>G</given-names> </name><name name-style="western"><surname>Weng</surname><given-names>C</given-names> </name></person-group><article-title>Rare disease diagnosis using knowledge guided retrieval augmentation for ChatGPT</article-title><source>J Biomed Inform</source><year>2024</year><month>09</month><volume>157</volume><fpage>104702</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2024.104702</pub-id><pub-id pub-id-type="medline">39084480</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alexandrou</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kumar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Mahtani</surname><given-names>AU</given-names> </name><etal/></person-group><article-title>Performance of large language models on the acute coronary syndrome guidelines using retrieval-augmented generation</article-title><source>JACC Cardiovasc Interv</source><year>2025</year><month>10</month><day>27</day><volume>18</volume><issue>20</issue><fpage>2458</fpage><lpage>2467</lpage><pub-id pub-id-type="doi">10.1016/j.jcin.2025.08.019</pub-id><pub-id pub-id-type="medline">41161918</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aminan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Darnell</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Delsoz</surname><given-names>M</given-names> </name><etal/></person-group><article-title>GlaucoRAG: a retrieval-augmented large language model for expert-level glaucoma assessment</article-title><source>medRxiv</source><year>2025</year><month>07</month><day>7</day><pub-id pub-id-type="doi">10.1101/2025.07.03.25330805</pub-id><pub-id pub-id-type="medline">40672509</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>R</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zheng</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>C</given-names> </name></person-group><article-title>Enhancing treatment decision-making for low back pain: a novel framework integrating large language models with retrieval-augmented generation technology</article-title><source>Front Med</source><year>2025</year><volume>12</volume><fpage>1599241</fpage><pub-id pub-id-type="doi">10.3389/fmed.2025.1599241</pub-id><pub-id pub-id-type="medline">40438365</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cremaschi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ditolve</surname><given-names>D</given-names> </name><name name-style="western"><surname>Curcio</surname><given-names>C</given-names> </name><name name-style="western"><surname>Panzeri</surname><given-names>A</given-names> </name><name name-style="western"><surname>Spoto</surname><given-names>A</given-names> </name><name name-style="western"><surname>Maurino</surname><given-names>A</given-names> </name></person-group><article-title>Decoding the mind: a RAG-LLM on ICD-11 for decision support in psychology</article-title><source>Expert Syst Appl</source><year>2025</year><month>06</month><volume>279</volume><fpage>127191</fpage><pub-id pub-id-type="doi">10.1016/j.eswa.2025.127191</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fink</surname><given-names>A</given-names> </name><name name-style="western"><surname>Nattenm&#x00FC;ller</surname><given-names>J</given-names> </name><name name-style="western"><surname>Rau</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Retrieval-augmented generation improves precision and trust of a GPT-4 model for emergency radiology diagnosis and classification: a proof-of-concept study</article-title><source>Eur Radiol</source><year>2025</year><month>08</month><volume>35</volume><issue>8</issue><fpage>5091</fpage><lpage>5098</lpage><pub-id pub-id-type="doi">10.1007/s00330-025-11445-z</pub-id><pub-id pub-id-type="medline">39953150</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gaber</surname><given-names>F</given-names> </name><name name-style="western"><surname>Shaik</surname><given-names>M</given-names> </name><name name-style="western"><surname>Allega</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Evaluating large language model workflows in clinical decision support for triage and referral and diagnosis</article-title><source>NPJ Digit Med</source><year>2025</year><month>05</month><day>9</day><volume>8</volume><issue>1</issue><fpage>263</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01684-1</pub-id><pub-id pub-id-type="medline">40346344</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Noll</surname><given-names>R</given-names> </name><name name-style="western"><surname>Windschmitt</surname><given-names>J</given-names> </name><name name-style="western"><surname>Hofmann</surname><given-names>E</given-names> </name><name name-style="western"><surname>Bergmann</surname><given-names>N</given-names> </name><name name-style="western"><surname>Schaaf</surname><given-names>J</given-names> </name></person-group><article-title>Retrieval-augmented generation for medical decision-making in emergency care</article-title><source>Annu Int Conf IEEE Eng Med Biol Soc</source><year>2025</year><month>07</month><volume>2025</volume><fpage>1</fpage><lpage>7</lpage><pub-id pub-id-type="doi">10.1109/EMBC58623.2025.11253463</pub-id><pub-id pub-id-type="medline">41337300</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ong</surname><given-names>JCL</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>L</given-names> </name><name name-style="western"><surname>Elangovan</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Large language model as clinical decision support system augments medication safety in 16 clinical specialties</article-title><source>Cell Rep Med</source><year>2025</year><month>10</month><day>21</day><volume>6</volume><issue>10</issue><fpage>102323</fpage><pub-id pub-id-type="doi">10.1016/j.xcrm.2025.102323</pub-id><pub-id pub-id-type="medline">40997804</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ozmen</surname><given-names>BB</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>N</given-names> </name><name name-style="western"><surname>Shah</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Development of a novel artificial intelligence clinical decision support tool for hand surgery: HandRAG</article-title><source>J Hand Microsurg</source><year>2025</year><month>07</month><volume>17</volume><issue>4</issue><fpage>100293</fpage><pub-id pub-id-type="doi">10.1016/j.jham.2025.100293</pub-id><pub-id pub-id-type="medline">40606653</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ozmen</surname><given-names>BB</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>N</given-names> </name><name name-style="western"><surname>Shah</surname><given-names>K</given-names> </name><etal/></person-group><article-title>MicroRAG: development of a novel artificial intelligence retrieval-augmented generation model for microsurgery clinical decision support</article-title><source>Microsurgery</source><year>2025</year><month>12</month><volume>45</volume><issue>8</issue><fpage>e70138</fpage><pub-id pub-id-type="doi">10.1002/micr.70138</pub-id><pub-id pub-id-type="medline">41235700</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thaker</surname><given-names>NG</given-names> </name><name name-style="western"><surname>Redjal</surname><given-names>N</given-names> </name><name name-style="western"><surname>Dicker</surname><given-names>A</given-names> </name><etal/></person-group><article-title>RadOncRAG: a novel retrieval-augmented generation framework improves large language model benchmark performance in radiation oncology</article-title><source>JCO Clin Cancer Inform</source><year>2025</year><month>11</month><volume>9</volume><fpage>e2500220</fpage><pub-id pub-id-type="doi">10.1200/CCI-25-00220</pub-id><pub-id pub-id-type="medline">41237352</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tung</surname><given-names>JYM</given-names> </name><name name-style="western"><surname>Le</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Yao</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Performance of retrieval-augmented generation large language models in guideline-concordant prostate-specific antigen testing: comparative study with junior clinicians</article-title><source>J Med Internet Res</source><year>2025</year><month>11</month><day>19</day><volume>27</volume><fpage>e78393</fpage><pub-id pub-id-type="doi">10.2196/78393</pub-id><pub-id pub-id-type="medline">41259800</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rewthamrongsris</surname><given-names>P</given-names> </name><name name-style="western"><surname>Thongchotchat</surname><given-names>V</given-names> </name><name name-style="western"><surname>Burapacheep</surname><given-names>J</given-names> </name><name name-style="western"><surname>Trachoo</surname><given-names>V</given-names> </name><name name-style="western"><surname>Khurshid</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Porntaveetus</surname><given-names>T</given-names> </name></person-group><article-title>Evaluating retrieval-augmented generation-large language models for infective endocarditis prophylaxis: clinical accuracy and efficiency</article-title><source>Int Dent J</source><year>2026</year><month>02</month><volume>76</volume><issue>1</issue><fpage>109344</fpage><pub-id pub-id-type="doi">10.1016/j.identj.2025.109344</pub-id><pub-id pub-id-type="medline">41453288</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>P</given-names> </name><name name-style="western"><surname>Patel</surname><given-names>A</given-names> </name><name name-style="western"><surname>Vallamchetla</surname><given-names>SK</given-names> </name><etal/></person-group><article-title>Optimizing retrieval-augmented generation (RAG) in clinical medicine: methods and performance evaluation</article-title><source>J Am Med Inform Assoc</source><year>2026</year><month>08</month><day>1</day><volume>33</volume><issue>8</issue><fpage>1436</fpage><lpage>1445</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocag056</pub-id><pub-id pub-id-type="medline">42243630</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gao</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Q</given-names> </name><etal/></person-group><article-title>Multimodal language model for jaw osteonecrosis diagnosis and treatment</article-title><source>J Dent Res</source><year>2025</year><month>11</month><volume>104</volume><issue>12</issue><fpage>1324</fpage><lpage>1332</lpage><pub-id pub-id-type="doi">10.1177/00220345251334575</pub-id><pub-id pub-id-type="medline">40552509</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gao</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>X</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Enhancing privacy-preserving deployable large language models for perioperative complication detection: a targeted strategy with LoRA fine-tuning</article-title><source>NPJ Digit Med</source><year>2025</year><month>12</month><day>13</day><volume>8</volume><issue>1</issue><fpage>773</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-02139-3</pub-id><pub-id pub-id-type="medline">41390570</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kwon</surname><given-names>M</given-names> </name><name name-style="western"><surname>Jang</surname><given-names>KJ</given-names> </name><name name-style="western"><surname>Baek</surname><given-names>SJ</given-names> </name><etal/></person-group><article-title>Ophtimus-V2-Tx: a compact domain-specific LLM for ophthalmic diagnosis and treatment planning</article-title><source>Sci Rep</source><year>2025</year><month>12</month><day>10</day><volume>15</volume><issue>1</issue><fpage>43532</fpage><pub-id pub-id-type="doi">10.1038/s41598-025-27410-1</pub-id><pub-id pub-id-type="medline">41372257</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lahiri</surname><given-names>AK</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>QV</given-names> </name></person-group><article-title>AlzheimerRAG: multimodal retrieval-augmented generation for clinical use cases</article-title><source>Mach Learn Knowl Extr</source><year>2025</year><volume>7</volume><issue>3</issue><fpage>89</fpage><pub-id pub-id-type="doi">10.3390/make7030089</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>PJ</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>A vision&#x2013;language foundation model for Alzheimer&#x2019;s disease diagnosis using MRI and clinical data</article-title><source>Alzheimers Dement</source><year>2025</year><month>12</month><volume>21</volume><issue>12</issue><fpage>e71029</fpage><pub-id pub-id-type="doi">10.1002/alz.71029</pub-id><pub-id pub-id-type="medline">41457461</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Song</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><name name-style="western"><surname>He</surname><given-names>F</given-names> </name><name name-style="western"><surname>Yin</surname><given-names>W</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>W</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>J</given-names> </name></person-group><article-title>Stroke diagnosis and prediction tool using ChatGLM: development and validation study</article-title><source>J Med Internet Res</source><year>2025</year><month>02</month><day>26</day><volume>27</volume><fpage>e67010</fpage><pub-id pub-id-type="doi">10.2196/67010</pub-id><pub-id pub-id-type="medline">40009850</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Li</surname><given-names>G</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Diagnosis assistant for liver cancer utilizing a large language model with three types of knowledge</article-title><source>Phys Med Biol</source><year>2025</year><month>05</month><day>2</day><volume>70</volume><issue>9</issue><pub-id pub-id-type="doi">10.1088/1361-6560/adcb17</pub-id><pub-id pub-id-type="medline">40203862</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>KC</given-names> </name><name name-style="western"><surname>Chew</surname><given-names>FY</given-names> </name><name name-style="western"><surname>Cheng</surname><given-names>KL</given-names> </name><etal/></person-group><article-title>Adaptive RAG-assisted MRI platform (ARAMP) for brain metastasis detection and reporting: a retrospective evaluation using post-contrast T1-weighted imaging</article-title><source>Bioengineering (Basel)</source><year>2025</year><month>06</month><day>26</day><volume>12</volume><issue>7</issue><fpage>698</fpage><pub-id pub-id-type="doi">10.3390/bioengineering12070698</pub-id><pub-id pub-id-type="medline">40722390</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cypko</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Salim</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Kumar</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Large language models with retrieval-augmented generation enhance expert modelling of Bayesian network for clinical decision support</article-title><source>Int J Comput Assist Radiol Surg</source><year>2026</year><month>02</month><volume>21</volume><issue>2</issue><fpage>211</fpage><lpage>222</lpage><pub-id pub-id-type="doi">10.1007/s11548-025-03524-9</pub-id><pub-id pub-id-type="medline">41182676</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hashjin</surname><given-names>NM</given-names> </name><name name-style="western"><surname>Amiri</surname><given-names>MH</given-names> </name><name name-style="western"><surname>Najafabadi</surname><given-names>MK</given-names> </name></person-group><article-title>DermaGPT a federated multimodal framework with a meta learned trust function for interpretable dermatology diagnostics</article-title><source>Sci Rep</source><year>2026</year><month>02</month><day>7</day><volume>16</volume><issue>1</issue><fpage>7959</fpage><pub-id pub-id-type="doi">10.1038/s41598-026-38715-0</pub-id><pub-id pub-id-type="medline">41654587</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>He</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name></person-group><article-title>Personalized diabetes treatment support using large language models fine-tuned on electronic health records: development and evaluation study</article-title><source>JMIR Form Res</source><year>2026</year><month>02</month><day>9</day><volume>10</volume><fpage>e71541</fpage><pub-id pub-id-type="doi">10.2196/71541</pub-id><pub-id pub-id-type="medline">41662664</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Anisuzzaman</surname><given-names>DM</given-names> </name><name name-style="western"><surname>Malins</surname><given-names>JG</given-names> </name><name name-style="western"><surname>Friedman</surname><given-names>PA</given-names> </name><name name-style="western"><surname>Attia</surname><given-names>ZI</given-names> </name></person-group><article-title>Fine-tuning large language models for specialized use cases</article-title><source>Mayo Clin Proc Digit Health</source><year>2025</year><month>03</month><volume>3</volume><issue>1</issue><fpage>100184</fpage><pub-id pub-id-type="doi">10.1016/j.mcpdig.2024.11.005</pub-id><pub-id pub-id-type="medline">40206998</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Search strategies, risk of bias, and characteristics of included studies.</p><media xlink:href="jmir_v28i1e104092_app1.docx" xlink:title="DOCX File, 158 KB"/></supplementary-material><supplementary-material id="app2"><label>Checklist 1</label><p>PRISMA checklist.</p><media xlink:href="jmir_v28i1e104092_app2.pdf" xlink:title="PDF File, 212 KB"/></supplementary-material></app-group></back></article>