<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="review-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e90046</article-id><article-id pub-id-type="doi">10.2196/90046</article-id><article-categories><subj-group subj-group-type="heading"><subject>Review</subject></subj-group></article-categories><title-group><article-title>Evaluation Methods for Inference-Time Retrieval-Augmented and Graph Retrieval-Augmented Large Language Models in Health Care: Scoping Review</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Zhao</surname><given-names>Yuhan</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1"/><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Miao</surname><given-names>Yiqun</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1"/><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Guo</surname><given-names>Rongrong</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Luo</surname><given-names>Yuan</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wang</surname><given-names>Huiying</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Wu</surname><given-names>Ying</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1"/></contrib></contrib-group><aff id="aff1"><institution>School of Nursing, Capital Medical University</institution><addr-line>No. 10 Xitoutiao, Youanmenwai, Fengtai District</addr-line><addr-line>Beijing</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Brini</surname><given-names>Stefano</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Chrimes</surname><given-names>Dillon</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Meleka</surname><given-names>Mark</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Borovic</surname><given-names>Mladen</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Ying Wu, PhD, School of Nursing, Capital Medical University, No. 10 Xitoutiao, Youanmenwai, Fengtai District, Beijing, 100069, China, 86 13910789837; <email>helenywu@vip.163.com</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>3</day><month>8</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e90046</elocation-id><history><date date-type="received"><day>20</day><month>12</month><year>2025</year></date><date date-type="rev-recd"><day>22</day><month>05</month><year>2026</year></date><date date-type="accepted"><day>01</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Yuhan Zhao, Yiqun Miao, Rongrong Guo, Yuan Luo, Huiying Wang, Ying Wu. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 3.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e90046"/><abstract><sec><title>Background</title><p>Inference-time retrieval augmentation is increasingly used to improve the traceability and verifiability of large language model (LLM) applications in health care. Evaluation practices for text-based retrieval-augmented generation (RAG) and graph-structured RAG (GraphRAG) systems remain heterogeneous, which limits comparison across studies and complicates judgments about clinical readiness.</p></sec><sec><title>Objective</title><p>This review mapped evaluation methods for inference-time retrieval-augmented and graph-structured retrieval-augmented LLM systems in health care and characterized how evaluation constructs are defined, operationalized, and reported across system layers and evaluation-setting categories.</p></sec><sec sec-type="methods"><title>Methods</title><p>We conducted a scoping review in accordance with PRISMA-ScR (Preferred Reporting Items for Systematic Reviews and Meta-Analyses extension for Scoping Reviews), with search reporting informed by PRISMA-S (PRISMA literature search extension). Searches were conducted through May 14, 2026, in PubMed (MEDLINE), Web of Science Core Collection, IEEE Xplore, ACM Digital Library, arXiv, and medRxiv, with backward and forward citation tracking of included studies. Eligible records described health care&#x2013;relevant LLM systems using inference-time RAG and reported at least 1 evaluation component. Data were charted on study characteristics, system design, retrieval-layer evaluation, evidence linkage, safety-related and GraphRAG-specific evaluation, and selected reporting and governance characteristics. We also constructed an evidence-and-gap map cross-classifying evaluation-setting categories with key evaluation domains.</p></sec><sec sec-type="results"><title>Results</title><p>A total of 157 studies met the inclusion criteria. Clinical question answering was the most frequently represented application (89/157, 56.7%), followed by clinical decision support (70/157, 44.6%). Most evaluations were conducted in offline-only settings (140/157, 89.2%), whereas 17/157 (10.8%) studies reported workflow-facing, prospective, or deployment-level evaluation. Independent retrieval-layer evaluation was reported in 47/157 (29.9%) studies. Grounding and faithfulness evaluation was reported in 41/157 (26.1%) studies, and fine-grained evidence verification was reported in 22/157 (14%) studies. Human evaluation was reported in 94/157 (59.9%) studies, but interrater reliability was reported in 26/94 (27.7%) studies. LLM-as-judge evaluation was reported in 41/157 (26.1%) studies, with bias-control measures reported in 15/41 (36.6%) studies. Formal safety-related evaluation was reported in 45/157 (28.7%) studies. Among 27 (17.2%) GraphRAG studies, intermediate-artifact evaluation was reported in 11/27 (40.7%) studies, and graph construction evaluation was reported in 6/27 (22.2%) studies. The evidence-and-gap map showed limited coverage of fine-grained verification, contradiction handling, safety evaluation, LLM-as-judge safeguards, GraphRAG construction evaluation, and GraphRAG intermediate-artifact evaluation in workflow-facing, prospective, or deployment-level settings.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Evaluation of health care RAG and GraphRAG systems has expanded rapidly, yet reporting and operational definitions remain inconsistent across evaluation layers. Current evidence remains concentrated in offline evaluation, with limited workflow-facing, prospective, or deployment-level assessment of retrieval quality, fine-grained evidence linkage, safety, LLM-as-judge safeguards, GraphRAG construction quality, and GraphRAG intermediate artifacts. This review maps these gaps across evaluation-setting categories and translates them into synthesis-informed evaluation considerations. These findings suggest that future evaluation may need to move beyond end-to-end benchmark performance toward more transparent, layer-specific, safety-oriented, and clinically contextualized assessment before workflow-facing implementation.</p></sec><sec><title>Trial Registration</title><p>OSF Registries mtf5x; https://osf.io/mtf5x/overview</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>natural language processing</kwd><kwd>artificial intelligence</kwd><kwd>retrieval-augmented generation</kwd><kwd>GraphRAG</kwd><kwd>information storage and retrieval</kwd><kwd>scoping review</kwd><kwd>evaluation studies as topic</kwd><kwd>hallucination</kwd><kwd>clinical decision support systems</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Rationale</title><p>Large language models (LLMs) are increasingly explored in health care for tasks such as clinical decision support, patient education, documentation, administrative workflows, and broader biomedical use cases [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref6">6</xref>]. However, study designs, evaluation end points, and reporting practices remain heterogeneous, and deployment in high-stakes medical settings is constrained by the tendency of these models to produce factually incorrect, internally inconsistent, or insufficiently supported statements [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. Such reliability limitations raise patient safety concerns and can undermine clinician trust in automated systems [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. Improving factual accuracy and enabling verifiable outputs have therefore become central objectives for medical artificial intelligence (AI) research [<xref ref-type="bibr" rid="ref11">11</xref>]. Accordingly, evaluation is not only a measure of technical performance but also a prerequisite for judging whether health care LLM systems are sufficiently transparent, safe, and interpretable before use in progressively more clinical or workflow-facing settings.</p><p>To mitigate factual unreliability, researchers have increasingly adopted inference-time retrieval-augmented generation (RAG) [<xref ref-type="bibr" rid="ref12">12</xref>]. This paradigm retrieves evidence from external knowledge sources such as clinical guidelines, biomedical literature, institutional protocols, and electronic health records (EHRs) during the generation process. By conditioning generation on retrieved context, these systems aim to improve accuracy and support traceability of outputs to source evidence [<xref ref-type="bibr" rid="ref13">13</xref>]. The approach includes text-based RAG and emerging graph-structured RAG (GraphRAG) [<xref ref-type="bibr" rid="ref14">14</xref>]. Many retrieval-augmented systems retrieve text using dense vector similarity or hybrid sparse-dense retrieval, whereas graph-based approaches use graph structure to condition inference-time evidence retrieval or organization for generation [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>]. Graph-based methods can also introduce unique intermediate artifacts such as retrieved paths, retrieved subgraphs, and graph community summaries, which create additional evaluation targets beyond end-to-end task performance. However, retrieval augmentation does not by itself ensure that retrieved evidence is relevant, that generated claims are faithfully supported by that evidence, that citation and source attributions are correct, or that outputs are clinically safe. RAG and GraphRAG systems therefore require evaluation methods that distinguish retrieval quality, evidence linkage, end-to-end output quality, safety-related behavior, and readiness for more clinically realistic settings.</p><p>Despite the rapid expansion of health care retrieval-augmented architectures, particularly since 2024, evaluation methodologies remain fragmented. While systematic and scoping reviews have mapped general LLM applications in clinical medicine, patient education, and broader health care settings [<xref ref-type="bibr" rid="ref2">2</xref>-<xref ref-type="bibr" rid="ref6">6</xref>], other reviews and guidance papers have focused more directly on testing, evaluation, and reporting of health care LLM applications [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. General RAG and GraphRAG reviews have also summarized retrieval-augmented architectures, retriever-generator integration, robustness issues, graph-based indexing, graph-guided retrieval, graph-enhanced generation, and emerging evaluation frameworks [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref18">18</xref>-<xref ref-type="bibr" rid="ref20">20</xref>]. These syntheses are important, but they have not primarily been designed to provide a layer-specific synthesis of health care evaluation methods for inference-time RAG and GraphRAG systems, including verification granularity, independent retrieval-layer evaluation, formal safety-related evaluation, and evaluation-setting context. Many studies emphasize end-to-end performance metrics, which can obscure the distinct contributions and failure modes of retrieval and generation components [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>].</p><p>RAG evaluation literature has further highlighted that retrieval-augmented systems pose distinctive evaluation challenges because their behavior depends on both retrieval and generation components as well as on dynamic external knowledge sources [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>]. A related review of medical RAG literature likewise suggests that research in this area has concentrated on technical implementations and clinical applications, whereas evaluation commonly relies on automated metrics or broad human judgments, with less explicit attention to bias and safety [<xref ref-type="bibr" rid="ref21">21</xref>]. In addition, it is often unclear to what extent evaluation protocols incorporate clinical validity, safety assessment, and testing in settings that approximate real clinical use [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. Evaluation methods for inference-time RAG and GraphRAG systems therefore warrant dedicated synthesis, particularly to clarify how layer-specific constructs are operationalized, how evaluation coverage varies across evaluation-setting categories and key evaluation domains, and where methodological gaps remain within health care domains. To our knowledge, no previous review has specifically mapped how evaluation methods for health care inference-time RAG and GraphRAG systems are operationalized across retrieval-layer evaluation, grounding and faithfulness evaluation, citation and source correctness evaluation, formal safety-related evaluation, human evaluation, automated metrics, LLM-as-judge evaluation, GraphRAG construction and intermediate-artifact evaluation, and evaluation-setting categories.</p></sec><sec id="s1-2"><title>Objectives</title><p>Given the heterogeneity in task definitions, data sources, and evaluation end points, we conducted a scoping review to map the evidence and characterize evaluation practices rather than to estimate pooled effects. The primary objective of this review was to systematically map evaluation methods used for inference-time RAG and GraphRAG systems in health care. Specifically, this review aimed to characterize evaluation constructs, measurement approaches, and study designs across system layers, including retrieval-layer evaluation, grounding and faithfulness evaluation, citation and source correctness evaluation, verification granularity, end-to-end task outcomes, human evaluation, automated metrics, LLM-as-judge evaluation and related safeguards, formal safety-related evaluation, GraphRAG construction evaluation, and GraphRAG-specific intermediate-artifact evaluation. By synthesizing these practices into a layer-specific taxonomy and mapping evaluation coverage across evaluation-setting categories and key evaluation domains, this review aimed to identify methodological gaps, differentiate the evaluation needs of text-based RAG and GraphRAG systems, and inform more transparent, setting-appropriate, and safety-oriented evaluation and reporting in future research.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Protocol and Registration</title><p>This scoping review was designed to systematically map and characterize evaluation methods used in health care LLM systems using inference-time RAG, with specific attention to both text-based RAG and GraphRAG approaches. The review was conducted in accordance with methodological guidance from the Joanna Briggs Institute for scoping reviews and reported in accordance with the PRISMA-ScR (Preferred Reporting Items for Systematic Reviews and Meta-Analyses Extension for Scoping Reviews) checklist (<xref ref-type="supplementary-material" rid="app3">Checklist 1</xref>) [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]. A protocol specifying the research questions, eligibility criteria, information sources, search strategy structure, screening procedures, data charting items, and coding framework was developed a priori and registered on the Open Science Framework (OSF) through OSF Registries before study selection and data charting were initiated (registration ID: MTF5X).</p><p>The objective was to identify and synthesize how evaluation constructs are defined and operationalized across system layers, to summarize study designs and evaluation modalities, to map evaluation coverage across descriptive evaluation-setting categories, and to characterize reporting practices relevant to the interpretation of evaluation methods, including governance indicators related to real patient data, deidentification, and ethics approval, exemption, or waiver. Specifically, this review sought to map how evaluation was performed at the retrieval-layer, grounding and faithfulness, evidence-verification, safety-related, GraphRAG-specific, and end-to-end system levels, how evaluation coverage varied across evaluation-setting categories, and how these evaluations were reported across heterogeneous health care settings. Given the expected heterogeneity in evaluation targets, end points, and study settings, a scoping review approach was selected to support comprehensive mapping and structured evidence synthesis rather than quantitative effect estimation.</p><p>This scoping review addressed 5 research questions aligned with the structure of the Results section.</p><p>First, what health care tasks and evaluation-setting categories are represented in studies evaluating inference-time RAG and GraphRAG LLM systems?</p><p>Second, what system design characteristics relevant to evaluation are reported, including knowledge source types, retrieval approaches, reranking stages, GraphRAG-related system features, and generator or evaluator model choices?</p><p>Third, how is retrieval-layer evaluation operationalized, including whether independent component-level evaluation is performed, which retrieval metrics are reported, how relevance labels or reference evidence are constructed, and how retrieval unit and granularity are specified?</p><p>Fourth, how are grounding and faithfulness outcomes evaluated, including how these constructs are operationalized, how outputs are linked to retrieved evidence, what units of analysis are used for verification, and whether citation and source correctness evaluation or conflict and contradiction handling is assessed?</p><p>Finally, what broader evaluation modalities and reporting practices are used, including human evaluation procedures, automated end point selection, LLM-as-judge evaluation procedures, formal safety-related evaluation, GraphRAG construction evaluation, GraphRAG intermediate-artifact evaluation, and selected governance-related elements?</p></sec><sec id="s2-2"><title>Ethical Considerations</title><p>Ethical approval was not required for this scoping review because it synthesized data from publicly available studies and did not involve human participants.</p></sec><sec id="s2-3"><title>Eligibility Criteria</title><p>Eligibility criteria were defined a priori using the Population, Concept, and Context (PCC) framework recommended for scoping reviews [<xref ref-type="bibr" rid="ref25">25</xref>]. In this review, the population corresponded to implemented health care&#x2013;relevant LLM systems, the concept to inference-time RAG and its evaluation, and the context to health care application settings. Records were eligible if they described a health care&#x2013;relevant system in which an LLM produced generative outputs and implemented RAG by retrieving external evidence during inference and incorporating retrieved material into generation, including text-based retrieval as well as GraphRAG approaches in which graph structure was used for inference-time evidence retrieval, organization, or assembly. Records were required to report at least 1 evaluation component (eg, retrieval performance, grounding and faithfulness, citation and source correctness, end-to-end task performance, human evaluation, formal safety-related evaluation, or GraphRAG-specific evaluation).</p><p>We excluded studies describing training-only knowledge injection without inference-time retrieval conditioning, retrieval systems without generative outputs, studies without empirical evaluation of an implemented system, narrative reviews, conference abstracts without accessible full text, and records for which full text could not be obtained. When multiple records described the same underlying study, a single record was retained for synthesis to avoid double counting. We preferentially retained the peer-reviewed version when it adequately described the evaluation methods; otherwise, the record providing the most complete description of the study design, system implementation, and evaluation procedures was retained as the primary synthesis record, with companion reports consulted as needed for clarification only.</p></sec><sec id="s2-4"><title>Information Sources</title><p>We searched PubMed (MEDLINE), Web of Science Core Collection, IEEE Xplore, and the ACM Digital Library. Searches were conducted on May 14, 2026, and were limited to records published or posted from January 1, 2024 to May 14, 2026. Searches were limited to English-language records. This period was selected to focus on contemporary evaluation practices from 2024 onward, alongside the broader uptake of frontier LLMs and specialized health care RAG frameworks. This window was intended to reflect recent evaluation paradigms for inference-time retrieval architectures in clinical domains rather than foundational natural language processing (NLP) tasks that predated their widespread health care use. To capture emerging work disseminated ahead of journal publication, we additionally searched arXiv and medRxiv using equivalent concept blocks adapted to platform-specific syntax and the same search cutoff. We also performed backward and forward citation tracking for all included studies using the same eligibility criteria to identify additional eligible records [<xref ref-type="bibr" rid="ref26">26</xref>].</p></sec><sec id="s2-5"><title>Search Strategy</title><p>The search strategy was developed iteratively by the review team and combined controlled vocabulary and free-text terms for (1) LLMs; (2) inference-time RAG, including GraphRAG approaches; (3) health care context; and (4) evaluation-related concepts. Controlled vocabulary was used where supported by the database, and syntax, fields, and limits were adapted to each platform. To align the search strategy with the review objective of synthesizing evaluation practices, the database queries included evaluation-related terms, including evaluation, benchmarking, metrics, grounding and faithfulness, hallucination, safety-related evaluation, and citation and source correctness, to improve identification of studies that explicitly reported evaluation methods, benchmarking designs, or evidence-verification procedures. Because the review focused specifically on evaluation practices, these terms were included to improve precision for studies reporting explicit evaluation methods; backward and forward citation tracking was used to mitigate the risk of missing eligible studies whose evaluation components were not captured in searchable titles, abstracts, or keywords. Literature searching and search reporting were conducted and documented in accordance with the PRISMA-S (PRISMA literature search extension) [<xref ref-type="bibr" rid="ref26">26</xref>]; an item-by-item PRISMA-S checklist, full database-specific search strategies, and supplementary search procedures are provided in <xref ref-type="supplementary-material" rid="app4">Checklist 2</xref> and <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. No study registries, structured website handsearching, contact-based supplementary identification, or formal search peer review was undertaken; these PRISMA-S items are reported explicitly in <xref ref-type="supplementary-material" rid="app4">Checklist 2</xref>.</p></sec><sec id="s2-6"><title>Selection of Sources of Evidence</title><p>Retrieved records were deduplicated in EndNote (Clarivate) and screened in a dedicated platform. Moreover, 2 reviewers (YZ and YM) independently screened titles and abstracts and then full texts of records classified as potentially eligible or uncertain. Reviewers completed a calibration exercise before full screening. Discrepancies were resolved through a consensus-seeking discussion between the 2 reviewers (YZ and YM); if consensus could not be reached, a third senior reviewer (YW) adjudicated the final inclusion decision. Reasons for exclusion at the full-text stage were recorded and are reported in the PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) flow diagram. Records identified through all information sources and citation tracking were screened using the same eligibility criteria and selection procedures.</p></sec><sec id="s2-7"><title>Data Charting Process</title><p>A standardized data-charting form was developed a priori to operationalize the review questions and support consistent charting of system characteristics, evaluation constructs, operational definitions, descriptive evaluation-setting categories, evaluation domains, and reporting elements. The form was refined iteratively through team discussion and was piloted on a prespecified sample of included studies to calibrate interpretation of fields and coding rules. Following calibration, one reviewer (YZ) charted data from all included studies and a second reviewer (YM) independently verified all entries against the full texts and available supplementary materials. Discrepancies were resolved through a consensus-seeking discussion. If consensus was not reached, a third senior reviewer (YW) adjudicated the final coding to maintain consistency across the review.</p><p>Charting was conducted at the study level and allowed multilabel coding when studies reported multiple tasks, knowledge sources, evaluation-setting categories, or evaluation modalities. Coding was based on full-text review and prespecified category definitions rather than title, abstract, or terminology alone. When information relevant to a field was explicitly reported, fields were coded according to the relevant category definitions. When information was not reported, fields were coded as &#x201C;not reported.&#x201D; Fields were coded as &#x201C;not applicable&#x201D; when the item was structurally irrelevant to the study design or evaluation approach. Ambiguous cases were resolved through full-text review and reviewer consensus rather than retained as a separate coding category. Coding decisions and adjudication notes were documented to preserve an audit trail.</p></sec><sec id="s2-8"><title>Data Items</title><p>Data were charted across five domains: (1) study characteristics and application settings, including application task categories and descriptive evaluation-setting categories; (2) system design characteristics relevant to evaluation, including knowledge source type, retrieval granularity, retrieval approach, reranking, generation model choice, evaluator model choice, and GraphRAG-related system features; (3) retrieval-layer evaluation, including retrieval metrics, relevance labeling procedures, reference evidence construction, and reporting of document-level or passage-level granularity; (4) grounding and faithfulness evaluation, including terminology used, operationalization approach, unit of analysis, fine-grained evidence verification, citation and source correctness, and handling of conflicting or contradictory evidence; and (5) evaluation modalities and reporting-related study characteristics, including human evaluation design and rater characteristics, interrater reliability (IRR) reporting, automated metrics, LLM-as-judge procedures and bias-control measures, formal safety-related evaluation, GraphRAG construction and intermediate-artifact evaluation, and governance indicators when explicitly described.</p></sec><sec id="s2-9"><title>Operational Definitions and Coding Framework</title><p>To support consistent and reproducible synthesis, we applied an explicit coding framework with operational definitions for evaluation layers, constructs, and study attributes. The framework was developed before full data charting, refined during calibration, and then applied uniformly across all included studies. Key operational definitions and primary coding units used in this review are summarized in <xref ref-type="table" rid="table1">Table 1</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Operational definitions and coding units used in this review.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Construct</td><td align="left" valign="bottom">Working definition in this review</td><td align="left" valign="bottom">Primary unit of assessment</td><td align="left" valign="bottom">Included under this construct</td><td align="left" valign="bottom">Not counted as this construct</td></tr></thead><tbody><tr><td align="left" valign="top">Retrieval-layer evaluation</td><td align="left" valign="top">Explicit assessment of the quality or performance of retrieved evidence independent of final generated outputs.</td><td align="left" valign="top">Retrieved item, ranked set, document, passage, node, path, or subgraph</td><td align="left" valign="top">Retrieval metrics, relevance judgments, structured assessment of retrieval quality, comparison of retrieved evidence sets</td><td align="left" valign="top">End-to-end task performance reported without explicit retrieval-layer assessment</td></tr><tr><td align="left" valign="top">Grounding and faithfulness evaluation</td><td align="left" valign="top">Assessment of whether generated answers, claims, statements, citations, or sources were supported by retrieved evidence or retrieved context, and whether generated outputs remained constrained by that evidence.</td><td align="left" valign="top">Response, answer, claim, statement, citation, source, or sentence</td><td align="left" valign="top">Evidence-support assessment, answer faithfulness assessment, source-supported output checking, claim verification against retrieved evidence, and assessment of unsupported additions relative to retrieved context</td><td align="left" valign="top">Factual correctness assessed only against an external reference standard without explicit linkage to retrieved evidence</td></tr><tr><td align="left" valign="top">Citation and source correctness evaluation</td><td align="left" valign="top">Assessment of whether cited or displayed sources existed, were accurate, and supported the corresponding answer content.</td><td align="left" valign="top">Citation, source, source excerpt, or linked answer segment</td><td align="left" valign="top">Source attribution checks, citation support checks, source existence checks, evidence-to-answer linkage assessment</td><td align="left" valign="top">Citation presence alone without verification that the cited or displayed source supports the answer content</td></tr><tr><td align="left" valign="top">Formal safety-related evaluation</td><td align="left" valign="top">Assessment in which safety, harm, unsafe recommendations, hallucination-related risk, undertriage, suicide risk, harmful content, or comparable safety outcomes were included as formal evaluation dimensions or end points.</td><td align="left" valign="top">Response, recommendation, decision, triage output, risk classification, or system behavior</td><td align="left" valign="top">Safety scores, harmfulness ratings, unsafe recommendation assessment, hallucination risk assessment, undertriage or overtriage harm evaluation, harmful content blocking</td><td align="left" valign="top">General discussion of safety risks without formal evaluation; accuracy or guideline concordance reported without a safety-related end point</td></tr><tr><td align="left" valign="top">Evaluation-setting category</td><td align="left" valign="top">Descriptive classification of evaluation conditions by their relationship to real-world health care use, rather than an ordinal maturity level.</td><td align="left" valign="top">Study or evaluation setting</td><td align="left" valign="top">Offline-only evaluation, simulated vignette or case evaluation, workflow pilot or user study, prospective clinical study, real-world deployment or postdeployment monitoring</td><td align="left" valign="top">General study setting description without enough information to classify the evaluation setting</td></tr><tr><td align="left" valign="top">GraphRAG<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> minimum criterion</td><td align="left" valign="top">Explicit use of graph structure at inference time to retrieve, organize, or assemble evidence that conditions generation.</td><td align="left" valign="top">System-level design feature</td><td align="left" valign="top">Retrieval of nodes, paths, subgraphs, graph-structured evidence assembly, graph-derived summaries, or graph-derived community summaries used at inference time</td><td align="left" valign="top">Knowledge graph use limited to background knowledge representation, training-time enrichment, or architecture description without graph-structured inference-time retrieval or evidence assembly</td></tr><tr><td align="left" valign="top">Graph construction evaluation</td><td align="left" valign="top">Explicit evaluation of graph construction quality or graph content quality.</td><td align="left" valign="top">Graph, node, edge, relation, triple, or graph-derived schema</td><td align="left" valign="top">Assessment of node correctness, edge correctness, relation quality, triple extraction quality, graph completeness, or graph construction accuracy</td><td align="left" valign="top">Reporting graph size, graph architecture, or graph construction workflow without evaluating graph quality</td></tr><tr><td align="left" valign="top">GraphRAG intermediate-artifact evaluation</td><td align="left" valign="top">Explicit evaluation of intermediate graph-related artifacts produced or used during graph-structured retrieval-augmented generation.</td><td align="left" valign="top">Node, edge, path, subgraph, graph-derived summary, retrieved graph context, or provenance-linked graph artifact</td><td align="left" valign="top">Assessment of retrieved nodes, retrieved paths, retrieved subgraphs, graph-derived summaries, graph-based context, or provenance-linked graph artifacts</td><td align="left" valign="top">End-to-end output evaluation of a GraphRAG system without assessment of graph-related intermediate artifacts</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>GraphRAG: graph-structured retrieval-augmented generation.</p></fn></table-wrap-foot></table-wrap><p>Evaluation targets were conceptualized as nonmutually exclusive layers reflecting the multicomponent nature of inference-time RAG systems, including retrieval-layer evaluation, grounding and faithfulness evaluation, and end-to-end task evaluation. In addition, studies were coded for cross-cutting evaluation modalities and reporting dimensions, including human evaluation, formal safety-related evaluation, efficiency and implementation-readiness indicators, graph construction evaluation, GraphRAG intermediate-artifact evaluation, and reporting and governance dimensions.</p><p>Retrieval-layer evaluation was coded as present only when a study reported an explicit assessment of retrieved evidence quality independent of final generated outputs. This included quantitative retrieval metrics, relevance judgments of retrieved items, or structured evaluation of retrieval performance. Studies that reported only end-to-end task performance or qualitative examples without explicit assessment of retrieved evidence were not coded as reporting retrieval-layer evaluation.</p><p>For the purposes of this review, grounding and faithfulness evaluation was defined as assessment of whether generated answers, claims, statements, citations, or sources were supported by retrieved evidence or retrieved context, and whether generated outputs remained constrained by that evidence. Because terminology varied across studies, grounding and faithfulness evaluations were identified based on described verification procedures rather than author-reported labels. Citation and source correctness, claim verification against retrieved evidence, and assessment of unsupported additions relative to retrieved context were included when they explicitly evaluated alignment between generated outputs and retrieved sources. Hallucination-related assessments were included under grounding and faithfulness only when unsupported content was evaluated with reference to retrieved evidence or retrieved context. This operationalization was informed by previous RAG evaluation literature that assesses answer faithfulness and related evidence-linked dimensions [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref20">20</xref>].</p><p>For each distinct grounding or evidence-linkage evaluation component, we recorded the most granular verification unit explicitly described, including response- or answer-level, claim- or statement-level, citation- or source-level, and sentence-level verification. Because individual studies could report multiple evaluation components at different verification units, the summary granularity categories were nonmutually exclusive. Fine-grained evidence verification was coded when studies reported claim-, statement-, citation-, source-, or sentence-level verification. When a relevant verification unit was not explicitly described, it was coded as &#x201C;not reported.&#x201D; Fields were coded as &#x201C;not applicable&#x201D; when the item was structurally irrelevant to the study design or evaluation approach. Ambiguous cases were resolved through full-text review and reviewer consensus rather than retained as a separate coding category.</p><p>Evaluation-setting categories were coded to describe the relationship between evaluation conditions and real-world health care use [<xref ref-type="bibr" rid="ref22">22</xref>]. These descriptive, nonmutually exclusive categories included offline-only evaluation, simulated vignette or case evaluation, workflow pilot or user study, prospective clinical study, and real-world deployment or postdeployment monitoring. These categories were not treated as ordinal maturity levels. When a study reported multiple evaluation-setting categories, all applicable categories were recorded.</p><p>GraphRAG systems were identified using a minimum criterion requiring explicit use of graph structure at inference time to retrieve, organize, or assemble evidence that conditioned generation [<xref ref-type="bibr" rid="ref27">27</xref>]. This included retrieval of nodes, paths, subgraphs, graph-derived summaries, or graph-derived community summaries. Studies that referenced the use of a knowledge graph without inference-time graph-structured retrieval or evidence assembly were not classified as GraphRAG. Classification was based on the reported inference-time role of graph structure in evidence retrieval or assembly rather than on the mere presence of a knowledge graph within the broader system architecture. Graph construction evaluation was coded separately when studies evaluated graph nodes, edges, relations, triples, or knowledge-graph construction quality. GraphRAG intermediate-artifact evaluation was coded when studies evaluated retrieved nodes, retrieved paths, retrieved subgraphs, graph-derived summaries, graph-based context, or provenance-linked graph artifacts.</p><p>When a study reported multiple tasks, multiple knowledge sources, or multiple evaluation methods, all applicable categories were coded. Information relevant to a field but not explicitly reported was coded as &#x201C;not reported.&#x201D; Items that were structurally irrelevant to a study design or evaluation approach were coded as &#x201C;not applicable.&#x201D; Coding decisions and adjudication notes were documented to preserve traceability [<xref ref-type="bibr" rid="ref11">11</xref>].</p></sec><sec id="s2-10"><title>Critical Appraisal of Individual Sources of Evidence</title><p>Consistent with scoping review methodology, we did not conduct formal methodological quality appraisal or risk-of-bias assessment [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]. This was because the aim of the review was to map the range and characteristics of evaluation practices rather than to estimate intervention effects or exclude studies on the basis of methodological quality.</p></sec><sec id="s2-11"><title>Reporting-Related Assessment</title><p>We conducted a structured assessment of evaluation reporting completeness to characterize transparency of evaluation practices and to support interpretation of methodological gaps across the evidence base. This assessment was descriptive rather than evaluative and was intended to characterize reporting transparency, not to rate methodological quality. Reporting assessment focused on whether key information necessary to interpret and compare evaluation results was explicitly described [<xref ref-type="bibr" rid="ref11">11</xref>].</p><p>For each included study, we recorded reporting of key system and evaluation elements, including (1) retrieval and knowledge source description, including corpus or source, retrieval unit and granularity, retriever, and reranking components; (2) grounding and faithfulness evaluation, including definition, operationalization, and unit of analysis; (3) human evaluation procedures, including rater background and IRR; (4) automated evaluation procedures, including automated metrics, LLM-as-judge evaluation, and bias-control measures specifically applied when an LLM was used as an evaluator; (5) formal safety-related evaluation; and (6) GraphRAG construction evaluation and GraphRAG intermediate-artifact evaluation when applicable.</p><p>For studies involving real patient data, we additionally recorded reporting of ethical governance elements, including deidentification procedures and institutional review board approval, exemption, or waiver [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref22">22</xref>]. No studies were excluded on the basis of reporting completeness or ethical reporting. Findings were synthesized descriptively to identify common reporting gaps and areas where standardized reporting guidance could improve transparency and comparability in future studies.</p></sec><sec id="s2-12"><title>Synthesis of Results</title><p>Given the heterogeneity in health care tasks, system architectures, evaluation targets, and outcome measures, quantitative meta-analysis was not appropriate. We therefore synthesized findings using descriptive statistics and narrative synthesis to map evaluation practices across included studies [<xref ref-type="bibr" rid="ref24">24</xref>]. For the structured charting domains, we summarized counts and proportions of studies reporting the corresponding evaluation construct or design element.</p><p>Studies were retained as the unit of analysis, and multilabel coding was permitted when a study reported multiple tasks, evaluation layers, knowledge sources, or evaluation-setting categories. Results were organized to reflect the layer-specific evaluation framework defined a priori. Synthesis addressed application settings and evaluation contexts; system design patterns relevant to evaluation; retrieval-layer evaluation; grounding and faithfulness evaluation; evaluation modalities, including human evaluation and LLM-as-judge evaluation; formal safety-related evaluation; GraphRAG construction and intermediate-artifact evaluation; descriptive evaluation-setting categories; and selected reporting and governance observations.</p><p>Where informative, cross-tabulations were used to support descriptive comparisons across task categories, evaluation-setting categories, evaluation domains, and system design features; no statistical hypothesis testing was performed. Evaluation-setting categories were used descriptively and were not analyzed as an ordinal maturity scale. Findings were summarized using tables and figures to support transparency and interpretability. Each included study contributed equally to the synthesis.</p><p>To visualize evaluation gaps, we constructed an evidence-and-gap map by cross-classifying descriptive evaluation-setting categories with key evaluation domains. Evaluation-setting categories included offline-only evaluation, simulated vignette or case evaluation, workflow pilot or user study, prospective clinical study, and real-world deployment or postdeployment monitoring. Evaluation domains included retrieval-layer evaluation, fine-grained evidence verification, citation and source correctness evaluation, conflict or contradiction handling, formal safety-related evaluation, human evaluation with IRR reporting, LLM-as-judge evaluation with bias-control measures, GraphRAG construction evaluation, and GraphRAG intermediate-artifact evaluation. Evaluation-setting categories and evaluation domains were coded as nonmutually exclusive; therefore, individual studies could contribute to more than one cell.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Search Results and Study Selection</title><p>A total of 157 studies met the inclusion criteria and were included in this scoping review [<xref ref-type="bibr" rid="ref28">28</xref>-<xref ref-type="bibr" rid="ref184">184</xref>]. The process of study identification, screening, and inclusion is summarized in the PRISMA-ScR flow diagram (<xref ref-type="fig" rid="figure1">Figure 1</xref>).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>PRISMA-ScR (Preferred Reporting Items for Systematic Reviews and Meta-Analyses extension for Scoping Reviews) flow diagram of the study selection process. This diagram outlines the systematic literature search and screening procedure conducted in accordance with PRISMA-ScR guidelines. It details the number of records identified from databases and preprint platforms (PubMed, Web of Science Core Collection, IEEE Xplore, ACM Digital Library, arXiv, and medRxiv), the number of duplicates removed, and the stepwise exclusion reasons applied during screening and full-text eligibility assessment. A final total of 157 studies met the inclusion criteria for the scoping review. Searches were conducted through May 14, 2026. Backward and forward citation tracking was performed for all included studies. LLM: large language model.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e90046_fig01.png"/></fig></sec><sec id="s3-2"><title>Study Characteristics and Application Settings</title><p>The included studies (N=157 [<xref ref-type="bibr" rid="ref28">28</xref>-<xref ref-type="bibr" rid="ref184">184</xref>]) covered a broad range of health care tasks and evaluation contexts (<xref ref-type="table" rid="table2">Table 2</xref>; Table S1 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). Clinical question answering was the most frequently studied application, reported in 89 (56.7%) studies. Other commonly reported applications included clinical decision support tasks (n=70, 44.6%), patient or caregiver education (n=27, 17.2%), and summary or report generation (n=24, 15.3%). Task categories were not mutually exclusive, and some systems addressed multiple downstream tasks. Overall, the evidence base remained concentrated in question answering and decision-support applications, with smaller clusters of patient education, report generation, medical visual question answering, and administrative or operational support applications.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Characteristics of included studies and application settings (N=157). Percentages use all included studies (N=157) as the denominator. Application task categories, evaluation-setting categories, and knowledge-source categories were coded as multilabel categories; percentages therefore are not expected to sum to 100% within those blocks. Offline-only evaluation indicates the absence of workflow-facing, prospective, or deployment-level evaluation; studies could also be coded as simulated vignette or case evaluation when applicable. Evaluation-setting categories were descriptive, nonmutually exclusive categories rather than ordinal maturity levels. Full study-level coding is provided in Table S1 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Domain and item</td><td align="left" valign="bottom">Count, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Application task categories</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Clinical question answering</td><td align="left" valign="top">89 (56.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Clinical decision support</td><td align="left" valign="top">70 (44.6)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Patient or caregiver education</td><td align="left" valign="top">27 (17.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Summary or report generation</td><td align="left" valign="top">24 (15.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medical visual question answering</td><td align="left" valign="top">7 (4.5)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medical imaging</td><td align="left" valign="top">6 (3.8)</td></tr><tr><td align="left" valign="top" colspan="2">Evaluation settings</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Offline-only evaluation</td><td align="left" valign="top">140 (89.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Simulated vignette or case evaluation</td><td align="left" valign="top">37 (23.6)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Any workflow-facing, prospective, or deployment-level evaluation</td><td align="left" valign="top">17 (10.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Workflow pilot or user study</td><td align="left" valign="top">17 (10.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Prospective clinical study</td><td align="left" valign="top">2 (1.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Real-world deployment or postdeployment monitoring</td><td align="left" valign="top">3 (1.9)</td></tr><tr><td align="left" valign="top" colspan="2">Knowledge source categories</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Public or public-mixed knowledge sources</td><td align="left" valign="top">119 (75.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Private institutional knowledge sources, including mixed sources</td><td align="left" valign="top">27 (17.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Graph-structured knowledge sources</td><td align="left" valign="top">7 (4.5)</td></tr><tr><td align="left" valign="top" colspan="2">EHR<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>-related subset</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Retrieval corpus explicitly EHR-related</td><td align="left" valign="top">21 (13.4)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>EHR: electronic health record.</p></fn></table-wrap-foot></table-wrap><p>Across evaluation-setting categories, 140 (89.2%) studies reported offline-only evaluation and did not describe workflow pilots, prospective studies, or real-world deployment or postdeployment monitoring (<xref ref-type="table" rid="table2">Table 2</xref>; Table S1 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). Simulated vignette or case evaluation was reported in 37 (23.6%) studies. Only 17 (10.8%) studies reported at least 1 workflow-facing, prospective, or deployment-level evaluation setting. This included workflow pilot or user study (n=17, 10.8%), prospective clinical study (n=2, 1.3%), and real-world deployment or postdeployment monitoring (n=3, 1.9%). These evaluation-setting categories were descriptive and not mutually exclusive. <xref ref-type="fig" rid="figure2">Figure 2</xref> visualizes the relationship between application task category and evaluation-setting category.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Heat map of health care application tasks by evaluation-setting category. The heat map shows study counts across health care application tasks and evaluation-setting categories, with evaluation-setting categories ordered descriptively from offline-only evaluation to real-world deployment or postdeployment monitoring (offline-only evaluation, simulated vignette or case evaluation, workflow pilot or user study, prospective clinical study, and real-world deployment or postdeployment monitoring). Cell labels indicate the number of included studies in each task-setting combination. Because application task and evaluation-setting category were coded as nonmutually exclusive categories, individual studies could contribute to multiple cells; therefore, row and column totals do not sum to the 157-study corpus. Evaluation-setting categories were used descriptively and were not treated as ordinal maturity levels. Full study-level coding is provided in Table S1 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e90046_fig02.png"/></fig><p>Knowledge sources used for retrieval varied across studies (<xref ref-type="table" rid="table2">Table 2</xref>; Table S1 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). Public or public-mixed knowledge sources were reported in 119 (75.8%) studies, whereas 27 (17.2%) studies used private institutional sources either exclusively or in combination with other sources. Graph-structured knowledge sources were coded as the primary knowledge-source category in 7 (4.5%) studies. Retrieval corpora explicitly described as EHR-related were reported in 21 (13.4%) studies.</p></sec><sec id="s3-3"><title>RAG System Design Patterns Relevant to Evaluation</title><p>System architectures and retrieval configurations were heterogeneous. Dense or vector-based retrieval was commonly reported, often alongside hybrid sparse-dense retrieval, reranking, or domain-specific corpus construction. Systems meeting the prespecified GraphRAG minimum criterion were reported in 27 (17.2%) studies. This count reflects studies in which graph structure participated in inference-time retrieval, evidence organization, reasoning, or generation, rather than studies that merely referenced a knowledge graph as background knowledge.</p></sec><sec id="s3-4"><title>Evaluation-Method Coverage Across Included Studies</title><p>Evaluation-method coverage across the included studies is summarized in <xref ref-type="table" rid="table3">Table 3</xref>. Overall, end-to-end task performance evaluation was reported more consistently than layer-specific retrieval assessment, evidence-linkage verification, citation and source correctness evaluation, conflict or contradiction handling, GraphRAG-specific artifact evaluation, and implementation-facing evaluation indicators.</p><p><xref ref-type="table" rid="table3">Table 3</xref> provides the corresponding counts and percentages for the principal evaluation domains, GraphRAG indicators, and governance-related reporting elements.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Evaluation methods, graph-structured retrieval-augmented generation (GraphRAG) indicators, and governance-related reporting elements. Unless otherwise specified, percentages use all included studies (N=157) as the denominator. Interrater reliability uses the subgroup denominator of studies with human evaluation (n=94). Bias-control measures among large language model (LLM)&#x2013;as-judge studies use the subgroup denominator of studies using LLM-as-judge evaluation (n=41). GraphRAG intermediate-artifact evaluation and graph construction evaluation use the subgroup denominator of studies meeting the GraphRAG minimum criterion (n=27). Deidentification and institutional review board (IRB) reporting use the subgroup denominator of studies using real patient data (n=49). Fine-grained evidence verification includes claim-level, statement-level, citation-level, source-level, or sentence-level verification. Evidence-verification granularity categories were nonmutually exclusive because individual studies could report distinct evaluation components at different verification units. Data are aligned with Tables S2 and S3 in Multimedia Appendix 2.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Domain and item</td><td align="left" valign="bottom">Studies, n/N (%)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Retrieval-layer evaluation</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Explicit retrieval-layer evaluation</td><td align="left" valign="top">47/157 (29.9)</td></tr><tr><td align="left" valign="top" colspan="2">Grounding and evidence linkage</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Grounding and faithfulness evaluation</td><td align="left" valign="top">41/157 (26.1)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Fine-grained evidence verification</td><td align="left" valign="top">22/157 (14)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Citation and source correctness evaluation</td><td align="left" valign="top">12/157 (7.6)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Conflict or contradiction handling evaluation</td><td align="left" valign="top">9/157 (5.7)</td></tr><tr><td align="left" valign="top" colspan="2">Evidence-verification granularity</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No evidence-support verification</td><td align="left" valign="top">116/157 (73.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Answer-level verification</td><td align="left" valign="top">20/157 (12.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Claim-level or statement-level verification</td><td align="left" valign="top">17/157 (10.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Citation-level or source-level verification</td><td align="left" valign="top">21/157 (13.4)</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="2">Evaluation modalities</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Human evaluation used</td><td align="left" valign="top">94/157 (59.9)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Interrater reliability reported among studies with human evaluation</td><td align="left" valign="top">26/94 (27.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Automated metrics reported</td><td align="left" valign="top">117/157 (74.5)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LLM-as-judge evaluation used</td><td align="left" valign="top">41/157 (26.1)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content></td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Bias-control measures reported among studies using LLM-as-judge evaluation</td><td align="left" valign="top">15/41 (36.6)</td></tr><tr><td align="left" valign="top" colspan="2">Safety-related evaluation</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Formal safety-related evaluation</td><td align="left" valign="top">45/157 (28.7)</td></tr><tr><td align="left" valign="top" colspan="2">GraphRAG-specific indicators</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Studies meeting GraphRAG minimum criterion</td><td align="left" valign="top">27/157 (17.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GraphRAG intermediate-artifact evaluation reported</td><td align="left" valign="top">11/27 (40.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Graph construction evaluation reported</td><td align="left" valign="top">6/27 (22.2)</td></tr><tr><td align="left" valign="top" colspan="2">Governance among studies using real patient data</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Studies using real patient data</td><td align="left" valign="top">49/157 (31.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Deidentification reported among studies using real patient data</td><td align="left" valign="top">33/49 (67.3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>IRB approval, exemption, or waiver reported among studies using real patient data</td><td align="left" valign="top">23/49 (46.9)</td></tr></tbody></table></table-wrap></sec><sec id="s3-5"><title>Retrieval-Layer Evaluation</title><p>Independent retrieval-layer evaluation was reported in 47 (29.9%) studies (<xref ref-type="table" rid="table3">Table 3</xref>; Table S2 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). In the remaining 110 (70.1%) studies, retrieval-layer evaluation was not reported, and evaluation was conducted at the level of final generated outputs or through qualitative examples. Thus, although all included systems used inference-time RAG, only a subset reported retrieval as a separable evaluation layer.</p><p>Among the 47 studies reporting retrieval-layer evaluation, commonly reported metric families included precision-based metrics (n=23, 48.9%), recall metrics (n=32, 68.1%), mean reciprocal rank (n=8, 17%), and normalized discounted cumulative gain (n=6, 12.8%; Table S2 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). Reporting of retrieval-layer evaluation varied substantially across studies. Some studies reported standard information retrieval metrics, whereas others used context precision, context recall, retrieval accuracy, relevance judgments, or structured assessment of retrieved evidence. Studies that only compared final answer accuracy, final task performance, or qualitative examples without explicit assessment of retrieved evidence were not counted as reporting retrieval-layer evaluation.</p></sec><sec id="s3-6"><title>Evaluation of Grounding and Faithfulness</title><p>Grounding and faithfulness evaluation was reported in 41 (26.1%) studies (<xref ref-type="table" rid="table3">Table 3</xref>; Table S2 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). Operationalizations varied across studies. Citation and source correctness evaluation was reported in 12 (7.6%) studies, and fine-grained evidence verification was reported in 22 (14%) studies. Conflict or contradiction handling evaluation was reported in 9 (5.7%) studies. These categories were not mutually exclusive.</p><p>Studies also varied in the granularity at which evidence linkage was assessed (<xref ref-type="table" rid="table3">Table 3</xref>; Table S2 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). Answer-level verification was reported in 20 (12.7%) studies. Claim-level or statement-level verification was reported in 17 (10.8%) studies. Citation-level or source-level verification was reported in 21 (13.4%) studies. Overall, 116 of 157 studies (73.9%) reported no evidence-support verification. Verification units were therefore often broad, incompletely specified, or absent.</p></sec><sec id="s3-7"><title>Evaluation Modalities: Human and Automated End Points</title><p>Human evaluation was reported in 94 (59.9%) studies (<xref ref-type="table" rid="table3">Table 3</xref>; Table S3 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). IRR was reported in 26 (27.7%) of the human-evaluated studies.</p><p>Automated metrics were reported in 117 (74.5%) studies (<xref ref-type="table" rid="table3">Table 3</xref>; Table S3 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). These included task-performance metrics, retrieval metrics, lexical-overlap or semantic-similarity metrics, and automated evaluation frameworks, depending on the study design and evaluation target.</p><p>LLM-as-judge evaluation was reported in 41 (26.1%) studies. Among the 41 (26.1%) studies using LLM-as-judge evaluation, 15 (36.6%) studies reported at least 1 bias-control measure (<xref ref-type="table" rid="table3">Table 3</xref>; Table S3 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>).</p><p>Formal safety-related evaluation was reported in 45 (28.7%) studies. Safety was operationalized heterogeneously across studies, including explicit safety ratings, clinical risk or harm rubrics, medication and contraindication safety end points, harmful-content blocking, suicide risk stratification, triage harm indices, regulatory harm assessment, and hallucination-related risk evaluation. <xref ref-type="table" rid="table4">Table 4</xref> summarizes the primary safety-related operationalization family assigned to each study with formal safety-related evaluation. Methodological safeguards, such as IRR reporting and bias-control measures for LLM-as-judge evaluation, were less frequently reported than the use of evaluation end points themselves.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Primary operationalization families for formal safety-related evaluation among studies with formal safety-related assessment (n=45). Categories reflect the primary safety-related operationalization family assigned to each study with formal safety-related evaluation (n=45). Categories are mutually exclusive for this main-text summary, although individual studies could include secondary safety-related elements. Study-level coding is provided in Table S3 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. Percentages use studies with formal safety-related evaluation (n=45) as the denominator.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Primary operationalization family</td><td align="left" valign="bottom">Studies, n (%)</td><td align="left" valign="bottom">Operationalization and evaluation modality</td></tr></thead><tbody><tr><td align="left" valign="top">Explicit safety, harm, or clinical-risk scoring within expert or LLM<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup> evaluation rubrics</td><td align="left" valign="top">22 (48.9)</td><td align="left" valign="top">Broad safety scoring, harmfulness ratings, clinical-risk ratings, or safety dimensions embedded in expert, clinician, user, or LLM-as-judge evaluation rubrics.</td></tr><tr><td align="left" valign="top">Clinical management safety end points and harm consequences</td><td align="left" valign="top">10 (22.2)</td><td align="left" valign="top">Medication safety, drug contraindication screening, antibiotic or opioid safety, refusal or escalation behavior, undertriage or overtriage harm, medical or regulatory harm, or other task-specific clinical safety end points.</td></tr><tr><td align="left" valign="top">Hallucination, factuality, or evidence misalignment framed as safety</td><td align="left" valign="top">6 (13.3)</td><td align="left" valign="top">Hallucination, factuality, source suitability, unsupported clinical content, or evidence misalignment explicitly treated as a safety-relevant failure mode.</td></tr><tr><td align="left" valign="top">Harmful-content, crisis, adversarial, or misuse-safety safeguards</td><td align="left" valign="top">5 (11.1)</td><td align="left" valign="top">Harmful-content blocking, crisis safeguards, adversarial safety prompting, misuse-oriented testing, or safety monitoring for mental health or patient-facing systems.</td></tr><tr><td align="left" valign="top">Public-health misinformation and fact-checking safety evaluation</td><td align="left" valign="top">2 (4.4)</td><td align="left" valign="top">Fact-checking or misinformation evaluation framed as public-health risk mitigation.</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>LLM: large language model.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-8"><title>Exploratory Findings for GraphRAG-Specific Evaluation</title><p>In total, 27 studies met the prespecified minimum GraphRAG criterion and were included in GraphRAG-specific analyses (<xref ref-type="table" rid="table3">Table 3</xref>; Table S3 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). All 27 GraphRAG studies reported end-to-end output evaluation. GraphRAG intermediate-artifact evaluation was reported in 11 of the 27 (40.7%) GraphRAG studies. Graph construction evaluation was reported in 6 (22.2%) studies. Evaluation of retrieved paths, subgraphs, graph-derived summaries, or provenance-linked graph artifacts remained uncommon. Given the still limited number of studies meeting the minimum GraphRAG criterion, these findings should be interpreted as exploratory.</p></sec><sec id="s3-9"><title>Reporting, Governance, and Study-Level Audit Trail</title><p>Reporting-related details relevant to the interpretation of evaluation procedures varied across the included studies. Graph-specific reporting and governance indicators also varied across studies (<xref ref-type="table" rid="table3">Table 3</xref>; Table S3 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>). Study-level reporting of knowledge-source type, indexing unit, and retriever type was heterogeneous, and graph-specific evaluation of intermediate artifacts was uncommon even among studies meeting the minimum GraphRAG criterion. Real patient data use was reported in 49 (31.2%) studies. Among these studies, deidentification procedures were reported in 33 (67.3%) studies and institutional review board approval, exemption, or waiver was reported in 23 (46.9%) studies. Concise study-level core coding tables and the source data for <xref ref-type="fig" rid="figure3">Figure 3</xref> are provided in Tables S1-S4 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Evidence-and-gap map of evaluation domains by evaluation-setting category. The evidence-and-gap map cross-classifies descriptive evaluation-setting categories with key evaluation domains. Evaluation-setting categories include offline-only evaluation, simulated vignette or case evaluation, workflow pilot or user study, prospective clinical study, and real-world deployment or postdeployment monitoring. Evaluation domains include retrieval-layer evaluation, fine-grained evidence verification, citation and source correctness evaluation, conflict or contradiction handling, formal safety-related evaluation, human evaluation with interrater reliability reporting, LLM-as-judge evaluation with bias-control measures, GraphRAG construction evaluation, and GraphRAG intermediate-artifact evaluation. Circle area indicates the number of studies reporting the corresponding evaluation domain. Fill intensity indicates within-category coverage proportion, calculated as the number of studies in a given evaluation-setting category reporting the evaluation domain divided by the total number of studies in that category. Evaluation-setting categories and evaluation domains were coded as nonmutually exclusive; therefore, counts across rows or columns were not expected to sum to the total corpus. Source data for the map are provided in Table S4 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. GraphRAG: graph-structured retrieval-augmented generation; LLM: large language model.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e90046_fig03.png"/></fig></sec><sec id="s3-10"><title>Evidence and Gap Map by Evaluation-Setting Category</title><p><xref ref-type="fig" rid="figure3">Figure 3</xref> presents an evidence-and-gap map (Table S4 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>) cross-classifying descriptive evaluation-setting categories with key evaluation domains. The map shows that evaluation coverage was concentrated in offline-only and simulated case-based evaluation categories. Across workflow-facing and prospective settings, coverage was sparse for fine-grained evidence verification, conflict or contradiction handling, formal safety-related evaluation, IRR reporting among studies with human evaluation, judge bias-control measures for LLM-as-judge evaluation, GraphRAG construction evaluation, and GraphRAG intermediate-artifact evaluation. This pattern indicates that the literature has developed more extensively for offline and simulated case-based evaluation than for workflow-facing, safety-oriented, governance-aware, or graph-artifact-specific evaluation.</p></sec><sec id="s3-11"><title>Synthesis-Informed Evaluation Considerations by Intended Evaluation Setting</title><p>Based on the mapped evaluation gaps, <xref ref-type="table" rid="table5">Table 5</xref> presents synthesis-informed evaluation considerations by intended evaluation setting. This table is intended as a practice-oriented synthesis rather than an empirically validated checklist, implementation guideline, or ordinal maturity scale. Because the included studies were concentrated in offline-only and simulated evaluations, and because real-world deployment or postdeployment monitoring was sparsely represented, considerations for workflow-facing, prospective, and deployment-level settings should be interpreted as future-oriented evaluation considerations informed by the observed gaps and the implementation-relevant issues identified in this review.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Synthesis-informed evaluation considerations by intended evaluation setting. This table is a practice-oriented synthesis based on the observed evaluation gaps and implementation-relevant considerations. It is not an empirically validated reporting checklist, implementation guideline, or ordinal maturity scale. The elements should be interpreted as considerations that may be adapted to task risk, user group, knowledge source, and intended use context.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Evaluation setting and use context</td><td align="left" valign="bottom">Evaluation components to consider</td><td align="left" valign="bottom">Evidence expectations and common failure modes</td></tr></thead><tbody><tr><td align="left" valign="top">Offline-only evaluation: retrospective benchmark or held-out test set without workflow embedding</td><td align="left" valign="top">Retrieval-layer metrics; end-to-end task metrics; at least one evidence-linkage check at response or claim level; explicit reporting of corpus, retriever, and generator</td><td align="left" valign="top">Reference evidence or relevance labels; retrieval unit and granularity; prompts and configuration sufficient to interpret results. Common failures include retrieval miss, unsupported synthesis, and benchmark overfitting.</td></tr><tr><td align="left" valign="top">Simulated vignette or case evaluation: simulated clinician- or patient-facing scenarios</td><td align="left" valign="top">Offline elements plus scenario-based human review; claim- or citation-level verification when feasible; contradiction handling; at least one task-relevant safety-related end point</td><td align="left" valign="top">Expert-authored or expert-curated cases; adjudication rule or clinician rubric; explicit handling of conflicting evidence. Common failures include unsafe advice, weak abstention, and brittle behavior under conflicting evidence.</td></tr><tr><td align="left" valign="top">Workflow pilot or user study: interaction in a workflow-resembling environment with intended users</td><td align="left" valign="top">Simulated-setting elements plus usability and workflow outcomes; time burden or efficiency; user trust and calibration; escalation or handoff assessment</td><td align="left" valign="top">Clearly described user group; protocolized task flow; predefined safety oversight during pilot use. Common failures include workflow disruption, overtrust, hidden latency, and poor handoff to clinicians.</td></tr><tr><td align="left" valign="top">Prospective clinical study: prospective assessment in live or near-live care processes</td><td align="left" valign="top">Workflow-pilot elements plus predefined safety monitoring; subgroup analysis; governance documentation; protocolized human oversight</td><td align="left" valign="top">Prospective protocol; monitoring triggers; incident definitions; ethics and governance reporting. Common failures include consequential harm, performance heterogeneity, and inadequate monitoring thresholds.</td></tr><tr><td align="left" valign="top">Real-world deployment or postdeployment monitoring: operational deployment or postimplementation surveillance</td><td align="left" valign="top">Ongoing drift surveillance; incident logging; feedback loops; periodic re-audit of retrieval and evidence-linkage performance; version tracking</td><td align="left" valign="top">Monitoring cadence; trigger thresholds; rollback or escalation pathways; clear governance ownership. Common failures include performance drift, silent failure, configuration drift, and surveillance without remediation.</td></tr></tbody></table></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This scoping review mapped how inference-time RAG and GraphRAG systems in health care were evaluated across retrieval, evidence linkage, end-to-end performance, safety, reporting, and evaluation-setting categories. Across studies published or posted from 2024 to May 14, 2026, evaluation practices expanded rapidly while definitions, units of analysis, and reporting conventions remained fragmented. Evaluation coverage was concentrated in offline-only and simulated evaluation settings and was sparser in workflow-facing, prospective, and deployment-level settings, especially for retrieval-layer evaluation, fine-grained evidence verification, contradiction handling, formal safety-related evaluation, bias-control measures for LLM-as-judge evaluation, GraphRAG intermediate artifacts, and deployment-level monitoring. These findings support descriptive mapping of evaluation practices and suggest that clinical readiness claims are more interpretable when supported by layer-specific evidence [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref22">22</xref>].</p></sec><sec id="s4-2"><title>Comparison With Previous Work</title><p>Independent retrieval-layer evaluation was reported in only a minority of included studies, despite inference-time retrieval being a defining feature of all eligible systems. When retrieval is evaluated only through final outputs, output-level errors are more difficult to attribute to retrieval failures, evidence selection failures, or generation failures [<xref ref-type="bibr" rid="ref12">12</xref>]. This limits failure-mode localization and weakens comparative interpretation across RAG pipelines. The practical consequence is that end-to-end improvements can be misattributed to retrieval changes when they may primarily reflect prompt design, decoding constraints, answer formatting, or evaluator sensitivity. Conversely, apparently weak performance can be driven by corpus segmentation or retrieval unit choices rather than by limitations in generation. Without retrieval-layer evaluation and clear specification of retrieval unit and granularity, claims about the causal role of retrieval in improving factual reliability remain difficult to substantiate. This interpretation is consistent with broader RAG evaluation literature, which emphasizes that hybrid systems require layer-aware assessment because retrieval relevance, generation quality, and answer faithfulness are related but nonidentical targets [<xref ref-type="bibr" rid="ref18">18</xref>]. It also extends previous health care LLM reviews, which have documented fragmented practices but have focused less directly on component-level evaluation and failure-mode localization in retrieval-augmented systems [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>].</p><p>The literature frequently reported evaluations labeled as grounding, faithfulness, citation and source correctness, or hallucination assessment, but these terms often referred to partially overlapping end points. Clear construct boundaries matter because these end points are not interchangeable and can yield conflicting conclusions when treated as substitutes [<xref ref-type="bibr" rid="ref185">185</xref>]. Previous RAG and hallucination-evaluation literature has emphasized related constructs, but the health care studies mapped in this review often operationalized them with limited granularity or inconsistent terminology. Grounding concerns whether generated claims are supported by retrieved evidence available at inference time. Faithfulness concerns whether the response remains appropriately constrained by retrieved evidence, without introducing unsupported elaborations. Factuality concerns correctness with respect to an external reference standard that may include information not present in retrieved context. Citation and source correctness concerns whether cited sources actually support the local claims they are attached to, including placement, specificity, and claim-to-source alignment. Recent scholarship supports this distinction, showing that citation presence or citation plausibility is not equivalent to citation and source correctness, because apparently correct citations may still reflect post hoc rationalization rather than genuine evidence use [<xref ref-type="bibr" rid="ref186">186</xref>]. Hallucination rubrics vary widely and may mix grounding violations, factuality errors, omissions, and miscalibrated certainty depending on rubric design [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref185">185</xref>]. Fine-grained evidence verification is therefore important because response-level assessment can mask localized unsupported claims, whereas claim-level, citation-level, or sentence-level verification can make evidence-linkage failures more visible [<xref ref-type="bibr" rid="ref187">187</xref>].</p><p>Graph-structured systems were evaluated primarily through end-to-end output outcomes, while evaluation of intermediate artifacts, such as graph construction validity, retrieved paths or subgraphs, graph-derived summaries, and provenance-related interpretability, remained less consistently reported. This limits the ability to substantiate the primary motivations for graph structure, particularly claims about improved traceability and interpretability. The practical consequence is that graph-structured approaches can be judged using end points that do not test their stated advantages, which weakens interpretation of when graph structure provides measurable benefits and which graph components contribute to improvements or failures. Because a limited but expanding subset of studies met the minimum GraphRAG criterion, these findings should be interpreted as exploratory rather than definitive. When interpretability or provenance is presented as a rationale for graph-structured RAG, intermediate-artifact evaluation may help interpret the added value of GraphRAG beyond conventional RAG pipelines [<xref ref-type="bibr" rid="ref15">15</xref>].</p><p>Evaluation-setting context is central to interpreting health care evaluation evidence. Most studies evaluated systems in offline-only settings, with vignette- or case-based testing used as an intermediate step in some work. Offline evaluation is essential for controlled iteration, yet it can underrepresent workflow constraints, incomplete context, time pressure, and local practice variation, all of which directly shape reliability and clinician reliance. This observation aligns with broader health care LLM reviews finding limited high-realism prospective evidence [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>]. Previous health care AI evaluation guidance has similarly emphasized staged evaluation before clinical implementation, including escalation, oversight, governance, and monitoring requirements as systems move closer to clinical use [<xref ref-type="bibr" rid="ref23">23</xref>]. In this review, the evidence-and-gap map showed sparse workflow-facing, prospective, and deployment-level coverage for fine-grained evidence verification, contradiction handling, safety-related assessment, human-evaluation reliability reporting, bias-control measures for LLM-as-judge evaluation, and GraphRAG intermediate-artifact evaluation. This pattern supports aligning evaluation expectations with the intended use context. Early-stage systems may reasonably begin with retrieval-layer testing, end-to-end task metrics, and evidence-linkage checks, whereas workflow-facing systems warrant added attention to safety monitoring, contradiction handling, escalation pathways, user interaction, and governance.</p><p>Human evaluation was reported in over half of studies, often involving clinicians or domain experts. However, IRR was reported in only a minority of human-evaluated studies. Limited reporting of rater expertise, training, and reliability restricts interpretation of end points such as usefulness, appropriateness, and safety, which are constructs sensitive to rubric framing and rater background. Without reliability evidence, apparent differences between systems may reflect measurement instability rather than robust behavioral differences [<xref ref-type="bibr" rid="ref188">188</xref>]. LLM-as-judge evaluation was used in a subset of studies, yet bias-control measures for LLM-as-judge evaluation were reported in fewer than half of those studies [<xref ref-type="bibr" rid="ref189">189</xref>]. This pattern suggests automated evaluation is being adopted for scalability with variable reporting of safeguards. Validity threats include prompt sensitivity [<xref ref-type="bibr" rid="ref190">190</xref>], judge drift, correlated model errors, and reward hacking [<xref ref-type="bibr" rid="ref191">191</xref>,<xref ref-type="bibr" rid="ref192">192</xref>]. Safety-related evaluation also requires explicit operationalization because grounded or factually accurate responses can still be unsafe when they omit warnings, express inappropriate certainty, ignore patient-specific constraints, or fail under conflicting evidence [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref193">193</xref>-<xref ref-type="bibr" rid="ref196">196</xref>].</p></sec><sec id="s4-3"><title>Implications for Evaluation and Reporting</title><p>Reporting of implementation-relevant study details varied across the included studies. In retrieval-augmented systems, incomplete reporting is particularly consequential because system behavior depends on corpus provenance and versioning, chunking policy, embedding and retrieval configurations, reranking settings, and citation policies [<xref ref-type="bibr" rid="ref197">197</xref>]. A concrete consequence is that nominally similar systems may behave differently due to unreported corpus or indexing differences, while readers attribute differences to model choice or retrieval method. Improving reporting transparency therefore requires emphasizing RAG-specific determinants of behavior as first-class reporting items rather than as implementation details [<xref ref-type="bibr" rid="ref8">8</xref>]. This observation strongly aligns with the movement toward LLM-specific reporting standards in biomedicine, such as TRIPOD-LLM [<xref ref-type="bibr" rid="ref8">8</xref>], and emerging governance frameworks emphasizing continuous drift monitoring [<xref ref-type="bibr" rid="ref198">198</xref>].</p><p>Based on the descriptive synthesis and the evidence-and-gap map, these findings point to several practice-oriented implications for future evaluation and reporting. Retrieval and generation may be more interpretable when evaluated as separable layers, particularly when retrieval is claimed to improve performance. Grounding, faithfulness, and citation and source correctness are more interpretable when explicitly defined and assessed at a stated verification unit. Contradiction handling, uncertainty communication, and formal safety-related evaluation are particularly relevant when systems are positioned for clinical or workflow-facing use. For human evaluation and LLM-as-judge evaluation, reporting safeguards that support measurement validity may improve interpretability, including rater expertise, IRR, evaluator identity, prompts or rubrics, calibration, and sensitivity analyses. Graph-structured systems should consider GraphRAG construction evaluation and intermediate-artifact evaluation when interpretability, provenance, or graph-based evidence organization is presented as a key contribution. These priorities are consistent with emerging reporting guidance for LLM studies, RAG failure-mode analyses, health care AI evaluation frameworks, and literature on LLM-as-judge evaluation and medical safety evaluation [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref189">189</xref>-<xref ref-type="bibr" rid="ref197">197</xref>].</p><p>The synthesis-informed evaluation considerations proposed in this review should therefore be interpreted as a practice-oriented synthesis to support more transparent and setting-appropriate evaluation, rather than as a mandatory checklist or an ordinal maturity scale. These considerations are intended to help align evaluation intensity with intended use context, while allowing task- and context-specific adaptation [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref197">197</xref>,<xref ref-type="bibr" rid="ref198">198</xref>].</p></sec><sec id="s4-4"><title>Limitations</title><p>This scoping review has limitations. First, the review focused on contemporary studies from 2024 onward, with searches conducted through May 14, 2026; earlier foundational work may therefore be underrepresented. The review was limited to English-language records, which may have excluded relevant studies reported in other languages. Second, consistent with scoping review methodology, the synthesis maps reported practices rather than estimating pooled effects, and formal risk-of-bias appraisal lay outside the review scope [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]. Third, findings depend on reporting completeness; procedures performed but left undescribed may be underestimated, and classification necessarily depended on study descriptions; information relevant to a field but not explicitly described was coded as not reported, and items structurally irrelevant to a study design or evaluation approach were coded as not applicable. Fourth, heterogeneity in tasks, corpora, including private and EHR-derived resources, and evaluation end points limited quantitative synthesis and supported the descriptive nature of the evidence-and-gap map. Finally, the identification of GraphRAG systems relied on prespecified operational criteria; studies may be classified differently under alternative definitions that use broader or narrower criteria for graph-structured inference-time retrieval and evidence assembly.</p></sec><sec id="s4-5"><title>Conclusions</title><p>This scoping review contributes to health care generative AI evaluation by mapping how RAG and GraphRAG systems are evaluated across retrieval, evidence linkage, safety, reporting, GraphRAG-specific artifacts, and evaluation-setting categories. Unlike previous reviews that broadly survey health care LLM applications or general NLP benchmarks, this study examines how evaluation is defined, operationalized, and reported within RAG and GraphRAG systems. The evidence-and-gap map shows areas of concentrated coverage and clinically relevant gaps, while the synthesis-informed evaluation considerations translate these gaps into practice-oriented evaluation considerations. These findings may inform more transparent, layer-specific, and safety-oriented evaluation planning for systems being considered for workflow-facing implementation [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref193">193</xref>,<xref ref-type="bibr" rid="ref198">198</xref>].</p></sec></sec></body><back><ack><p>No generative artificial intelligence tools were used in the preparation of this manuscript.</p></ack><notes><sec><title>Funding</title><p>This research was supported by the Key Program of the National Natural Science Foundation of China (72034005) and the Specialised Organised Research Project of the Chinese Institutes for Medical Research (CX23YZ02). The funders had no role in the design, conduct, analysis, interpretation, or reporting of this scoping review.</p></sec><sec><title>Data Availability</title><p>The study-level core coding tables used to generate the manuscript tables and figures are provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. Additional audit materials are retained by the authors and can be made available by the corresponding author upon reasonable request.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: YZ, YM, YW</p><p>Data curation: YZ, YM, RG, HW</p><p>Formal analysis: YZ</p><p>Funding acquisition: YW</p><p>Investigation: YZ, YM, RG, HW</p><p>Methodology: YZ, YM, YL, YW</p><p>Project administration: YW</p><p>Resources: YL</p><p>Software: YZ, YM</p><p>Supervision: YW</p><p>Validation: YL, RG, YW</p><p>Visualization: YL, HW</p><p>Writing &#x2013; original draft: YZ</p><p>Writing &#x2013; review &#x0026; editing: YM, RG, HW, YW</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AI</term><def><p>artificial intelligence</p></def></def-item><def-item><term id="abb2">EHR</term><def><p>electronic health record</p></def></def-item><def-item><term id="abb3">GraphRAG</term><def><p>graph-structured retrieval-augmented generation</p></def></def-item><def-item><term id="abb4">IRR</term><def><p>interrater reliability</p></def></def-item><def-item><term id="abb5">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb6">NLP</term><def><p>natural language processing</p></def></def-item><def-item><term id="abb7">OSF</term><def><p>Open Science Framework</p></def></def-item><def-item><term id="abb8">PCC</term><def><p>Population, Concept, and Context</p></def></def-item><def-item><term id="abb9">PRISMA</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses</p></def></def-item><def-item><term id="abb10">PRISMA-S</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses literature search extension</p></def></def-item><def-item><term id="abb11">PRISMA-ScR</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses extension for Scoping Reviews</p></def></def-item><def-item><term id="abb12">RAG</term><def><p>retrieval-augmented generation</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bedi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Orr-Ewing</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Testing and evaluation of health care applications of large language models: a systematic review</article-title><source>JAMA</source><year>2025</year><month>01</month><day>28</day><volume>333</volume><issue>4</issue><fpage>319</fpage><lpage>328</lpage><pub-id pub-id-type="doi">10.1001/jama.2024.21700</pub-id><pub-id pub-id-type="medline">39405325</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liang</surname><given-names>EN</given-names> </name><name name-style="western"><surname>Pei</surname><given-names>S</given-names> </name><name name-style="western"><surname>Staibano</surname><given-names>P</given-names> </name><name name-style="western"><surname>van der Woerd</surname><given-names>B</given-names> </name></person-group><article-title>Clinical applications of large language models in medicine and surgery: a scoping review</article-title><source>J Int Med Res</source><year>2025</year><month>07</month><volume>53</volume><issue>7</issue><fpage>3000605251347556</fpage><pub-id pub-id-type="doi">10.1177/03000605251347556</pub-id><pub-id pub-id-type="medline">40615349</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aydin</surname><given-names>S</given-names> </name><name name-style="western"><surname>Karabacak</surname><given-names>M</given-names> </name><name name-style="western"><surname>Vlachos</surname><given-names>V</given-names> </name><name name-style="western"><surname>Margetis</surname><given-names>K</given-names> </name></person-group><article-title>Large language models in patient education: a scoping review of applications in medicine</article-title><source>Front Med (Lausanne)</source><year>2024</year><volume>11</volume><fpage>1477898</fpage><pub-id pub-id-type="doi">10.3389/fmed.2024.1477898</pub-id><pub-id pub-id-type="medline">39534227</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Meng</surname><given-names>X</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>K</given-names> </name><etal/></person-group><article-title>The application of large language models in medicine: a scoping review</article-title><source>iScience</source><year>2024</year><month>05</month><day>17</day><volume>27</volume><issue>5</issue><fpage>109713</fpage><pub-id pub-id-type="doi">10.1016/j.isci.2024.109713</pub-id><pub-id pub-id-type="medline">38746668</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>SF</given-names> </name><name name-style="western"><surname>Alyakin</surname><given-names>A</given-names> </name><name name-style="western"><surname>Seas</surname><given-names>A</given-names> </name><etal/></person-group><article-title>LLM-assisted systematic review of large language models in clinical medicine</article-title><source>Nat Med</source><year>2026</year><month>03</month><volume>32</volume><issue>3</issue><fpage>1152</fpage><lpage>1159</lpage><pub-id pub-id-type="doi">10.1038/s41591-026-04229-5</pub-id><pub-id pub-id-type="medline">41776077</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shool</surname><given-names>S</given-names> </name><name name-style="western"><surname>Adimi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Saboori Amleshi</surname><given-names>R</given-names> </name><name name-style="western"><surname>Bitaraf</surname><given-names>E</given-names> </name><name name-style="western"><surname>Golpira</surname><given-names>R</given-names> </name><name name-style="western"><surname>Tara</surname><given-names>M</given-names> </name></person-group><article-title>A systematic review of large language model (LLM) evaluations in clinical medicine</article-title><source>BMC Med Inform Decis Mak</source><year>2025</year><month>03</month><day>7</day><volume>25</volume><issue>1</issue><fpage>117</fpage><pub-id pub-id-type="doi">10.1186/s12911-025-02954-4</pub-id><pub-id pub-id-type="medline">40055694</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Gottweis</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Toward expert-level medical question answering with large language models</article-title><source>Nat Med</source><year>2025</year><month>03</month><volume>31</volume><issue>3</issue><fpage>943</fpage><lpage>950</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03423-7</pub-id><pub-id pub-id-type="medline">39779926</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zhu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhuang</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Can we trust AI doctors? A survey of medical hallucination in large language and large vision-language models</article-title><conf-name>Findings of the Association for Computational Linguistics</conf-name><conf-date>Jul 27 to Aug 1, 2025</conf-date><conf-loc>Vienna, Austria</conf-loc><fpage>6748</fpage><lpage>6769</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.findings-acl.350</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abdelwanis</surname><given-names>M</given-names> </name><name name-style="western"><surname>Alarafati</surname><given-names>HK</given-names> </name><name name-style="western"><surname>Tammam</surname><given-names>MMS</given-names> </name><name name-style="western"><surname>Simsekler</surname><given-names>MCE</given-names> </name></person-group><article-title>Exploring the risks of automation bias in healthcare artificial intelligence applications: a bowtie analysis</article-title><source>Journal of Safety Science and Resilience</source><year>2024</year><month>12</month><volume>5</volume><issue>4</issue><fpage>460</fpage><lpage>469</lpage><pub-id pub-id-type="doi">10.1016/j.jnlssr.2024.06.001</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tun</surname><given-names>HM</given-names> </name><name name-style="western"><surname>Rahman</surname><given-names>HA</given-names> </name><name name-style="western"><surname>Naing</surname><given-names>L</given-names> </name><name name-style="western"><surname>Malik</surname><given-names>OA</given-names> </name></person-group><article-title>Trust in artificial intelligence-based clinical decision support systems among health care workers: systematic review</article-title><source>J Med Internet Res</source><year>2025</year><month>07</month><day>29</day><volume>27</volume><fpage>e69678</fpage><pub-id pub-id-type="doi">10.2196/69678</pub-id><pub-id pub-id-type="medline">40772775</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gallifant</surname><given-names>J</given-names> </name><name name-style="western"><surname>Afshar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ameen</surname><given-names>S</given-names> </name><etal/></person-group><article-title>The TRIPOD-LLM reporting guideline for studies using large language models</article-title><source>Nat Med</source><year>2025</year><month>01</month><volume>31</volume><issue>1</issue><fpage>60</fpage><lpage>69</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03425-5</pub-id><pub-id pub-id-type="medline">39779929</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Gao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Xiong</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Retrieval-augmented generation for large language models: a survey</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 18, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2312.10997</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Gao</surname><given-names>T</given-names> </name><name name-style="western"><surname>Yen</surname><given-names>H</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>D</given-names> </name></person-group><article-title>Enabling large language models to generate text with citations</article-title><conf-name>Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Dec 6-10, 2023</conf-date><pub-id pub-id-type="doi">10.18653/v1/2023.emnlp-main.398</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Edge</surname><given-names>D</given-names> </name><name name-style="western"><surname>Trinh</surname><given-names>H</given-names> </name><name name-style="western"><surname>Cheng</surname><given-names>N</given-names> </name><etal/></person-group><article-title>From local to global: a GraphRAG approach to query-focused summarization</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 24, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2404.16130</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Peng</surname><given-names>B</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Graph retrieval-augmented generation: a survey</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 15, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2408.08921</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Gupta</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ranjan</surname><given-names>R</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>SN</given-names> </name></person-group><article-title>A comprehensive survey of retrieval-augmented generation (RAG): evolution, current landscape and future directions</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 3, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2410.12837</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>X</given-names> </name><name name-style="western"><surname>Xiang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>He</surname><given-names>M</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>D</given-names> </name></person-group><article-title>Evaluating large language models and agents in healthcare: key challenges in clinical applications</article-title><source>Intelligent Medicine</source><year>2025</year><month>05</month><volume>5</volume><issue>2</issue><fpage>151</fpage><lpage>163</lpage><pub-id pub-id-type="doi">10.1016/j.imed.2025.03.002</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Yu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Gan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>K</given-names> </name><name name-style="western"><surname>Tong</surname><given-names>S</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Q</given-names> </name></person-group><article-title>Evaluation of retrieval-augmented generation: a survey</article-title><conf-name>CCF Conference on Big Data</conf-name><conf-date>Aug 9-11, 2024</conf-date><pub-id pub-id-type="doi">10.1007/978-981-96-1024-2_8</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Gan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Retrieval augmented generation evaluation in the era of large language models: a comprehensive survey</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 21, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2504.14891</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Saad-Falcon</surname><given-names>J</given-names> </name><name name-style="western"><surname>Khattab</surname><given-names>O</given-names> </name><name name-style="western"><surname>Potts</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zaharia</surname><given-names>M</given-names> </name></person-group><article-title>ARES: an automated evaluation framework for retrieval-augmented generation systems</article-title><conf-name>Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies</conf-name><conf-date>Jun 16-21, 2024</conf-date><pub-id pub-id-type="doi">10.18653/v1/2024.naacl-long.20</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>R</given-names> </name><name name-style="western"><surname>Wong</surname><given-names>MYH</given-names> </name><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Retrieval-augmented generation in medicine: a scoping review of technical implementations, clinical applications, and ethical considerations</article-title><source>arXiv</source><comment>Preprint posted online on  Nov 8, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2511.05901</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Reddy</surname><given-names>S</given-names> </name><name name-style="western"><surname>Rogers</surname><given-names>W</given-names> </name><name name-style="western"><surname>Makinen</surname><given-names>VP</given-names> </name><etal/></person-group><article-title>Evaluation framework to guide implementation of AI systems into healthcare settings</article-title><source>BMJ Health Care Inform</source><year>2021</year><month>10</month><volume>28</volume><issue>1</issue><fpage>e100444</fpage><pub-id pub-id-type="doi">10.1136/bmjhci-2021-100444</pub-id><pub-id pub-id-type="medline">34642177</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vasey</surname><given-names>B</given-names> </name><name name-style="western"><surname>Nagendran</surname><given-names>M</given-names> </name><name name-style="western"><surname>Campbell</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Reporting guideline for the early stage clinical evaluation of decision support systems driven by artificial intelligence: DECIDE-AI</article-title><source>BMJ</source><year>2022</year><month>05</month><day>18</day><volume>377</volume><fpage>e070904</fpage><pub-id pub-id-type="doi">10.1136/bmj-2022-070904</pub-id><pub-id pub-id-type="medline">35584845</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tricco</surname><given-names>AC</given-names> </name><name name-style="western"><surname>Lillie</surname><given-names>E</given-names> </name><name name-style="western"><surname>Zarin</surname><given-names>W</given-names> </name><etal/></person-group><article-title>PRISMA extension for Scoping Reviews (PRISMA-ScR): checklist and explanation</article-title><source>Ann Intern Med</source><year>2018</year><month>10</month><day>2</day><volume>169</volume><issue>7</issue><fpage>467</fpage><lpage>473</lpage><pub-id pub-id-type="doi">10.7326/M18-0850</pub-id><pub-id pub-id-type="medline">30178033</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Peters</surname><given-names>MDJ</given-names> </name><name name-style="western"><surname>Marnie</surname><given-names>C</given-names> </name><name name-style="western"><surname>Tricco</surname><given-names>AC</given-names> </name><etal/></person-group><article-title>Updated methodological guidance for the conduct of scoping reviews</article-title><source>JBI Evid Synth</source><year>2020</year><month>10</month><volume>18</volume><issue>10</issue><fpage>2119</fpage><lpage>2126</lpage><pub-id pub-id-type="doi">10.11124/JBIES-20-00167</pub-id><pub-id pub-id-type="medline">33038124</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rethlefsen</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Kirtley</surname><given-names>S</given-names> </name><name name-style="western"><surname>Waffenschmidt</surname><given-names>S</given-names> </name><etal/></person-group><article-title>PRISMA-S: an extension to the PRISMA statement for reporting literature searches in systematic reviews</article-title><source>Syst Rev</source><year>2021</year><month>01</month><day>26</day><volume>10</volume><issue>1</issue><fpage>39</fpage><pub-id pub-id-type="doi">10.1186/s13643-020-01542-z</pub-id><pub-id pub-id-type="medline">33499930</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Han</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Shomer</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Retrieval-augmented generation with graphs (GraphRAG)</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 31, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2501.00309</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Karim</surname><given-names>A</given-names> </name><name name-style="western"><surname>Uzuner</surname><given-names>O</given-names> </name></person-group><article-title>MasonNLP at MEDIQA-WV 2025: multimodal retrieval-augmented generation with large language models for medical VQA</article-title><access-date>2026-07-05</access-date><conf-name>Proceedings of the 7th Clinical Natural Language Processing Workshop</conf-name><conf-date>Oct 30, 2025</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2025.clinicalnlp-1.10">https://aclanthology.org/2025.clinicalnlp-1.10</ext-link></comment></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wada</surname><given-names>A</given-names> </name><name name-style="western"><surname>Tanaka</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Nishizawa</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Retrieval-augmented generation elevates local LLM quality in radiology contrast media consultation</article-title><source>NPJ Digit Med</source><year>2025</year><month>07</month><day>2</day><volume>8</volume><issue>1</issue><fpage>395</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01802-z</pub-id><pub-id pub-id-type="medline">40604147</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fink</surname><given-names>A</given-names> </name><name name-style="western"><surname>Nattenm&#x00FC;ller</surname><given-names>J</given-names> </name><name name-style="western"><surname>Rau</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Retrieval-augmented generation improves precision and trust of a GPT-4 model for emergency radiology diagnosis and classification: a proof-of-concept study</article-title><source>Eur Radiol</source><year>2025</year><month>08</month><volume>35</volume><issue>8</issue><fpage>5091</fpage><lpage>5098</lpage><pub-id pub-id-type="doi">10.1007/s00330-025-11445-z</pub-id><pub-id pub-id-type="medline">39953150</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kelly</surname><given-names>A</given-names> </name><name name-style="western"><surname>Noctor</surname><given-names>E</given-names> </name><name name-style="western"><surname>Ryan</surname><given-names>L</given-names> </name><name name-style="western"><surname>van de Ven</surname><given-names>P</given-names> </name></person-group><article-title>The effectiveness of a custom AI chatbot for type 2 diabetes mellitus health literacy: development and evaluation study</article-title><source>J Med Internet Res</source><year>2025</year><month>05</month><day>5</day><volume>27</volume><fpage>e70131</fpage><pub-id pub-id-type="doi">10.2196/70131</pub-id><pub-id pub-id-type="medline">40324160</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Su</surname><given-names>AY</given-names> </name><name name-style="western"><surname>Knebel</surname><given-names>A</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>AY</given-names> </name><etal/></person-group><article-title>Evaluation of retrieval-augmented generation and large language models in clinical guidelines for degenerative spine conditions</article-title><source>Eur Spine J</source><year>2026</year><month>03</month><volume>35</volume><issue>3</issue><fpage>1301</fpage><lpage>1310</lpage><pub-id pub-id-type="doi">10.1007/s00586-025-08994-8</pub-id><pub-id pub-id-type="medline">40619525</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ho</surname><given-names>CM</given-names> </name><name name-style="western"><surname>Guan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Mok</surname><given-names>PKL</given-names> </name><etal/></person-group><article-title>Development and validation of a large language model-powered chatbot for neurosurgery: mixed methods study on enhancing perioperative patient education</article-title><source>J Med Internet Res</source><year>2025</year><month>07</month><day>15</day><volume>27</volume><fpage>e74299</fpage><pub-id pub-id-type="doi">10.2196/74299</pub-id><pub-id pub-id-type="medline">40663377</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Gan</surname><given-names>C</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>B</given-names> </name><etal/></person-group><article-title>POLYRAG: integrating polyviews into retrieval-augmented generation for medical applications</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 21, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2504.14917</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tarabanis</surname><given-names>C</given-names> </name><name name-style="western"><surname>Khurshid</surname><given-names>S</given-names> </name><name name-style="western"><surname>Karamanou</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Cardiology knowledge assessment of retrieval-augmented open versus proprietary large language models</article-title><source>PLOS Digit Health</source><year>2026</year><month>03</month><volume>5</volume><issue>3</issue><fpage>e0001029</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0001029</pub-id><pub-id pub-id-type="medline">41818295</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>de Jesus</surname><given-names>DDR</given-names> </name><name name-style="western"><surname>de Souza J&#x00FA;nior</surname><given-names>AP</given-names> </name><name name-style="western"><surname>de Albergaria</surname><given-names>ET</given-names> </name><etal/></person-group><article-title>Enhanced LLM-supported instructions for medication use through retrieval-augmented generation</article-title><source>Comput Biol Med</source><year>2025</year><month>11</month><volume>198</volume><issue>Pt A</issue><fpage>111135</fpage><pub-id pub-id-type="doi">10.1016/j.compbiomed.2025.111135</pub-id><pub-id pub-id-type="medline">41077040</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Baur</surname><given-names>D</given-names> </name><name name-style="western"><surname>Ansorg</surname><given-names>J</given-names> </name><name name-style="western"><surname>Heyde</surname><given-names>CE</given-names> </name><name name-style="western"><surname>Voelker</surname><given-names>A</given-names> </name></person-group><article-title>Development and evaluation of a retrieval-augmented generation chatbot for orthopedic and trauma surgery patient education: mixed-methods study</article-title><source>JMIR AI</source><year>2025</year><month>10</month><day>23</day><volume>4</volume><fpage>e75262</fpage><pub-id pub-id-type="doi">10.2196/75262</pub-id><pub-id pub-id-type="medline">41134117</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Steybe</surname><given-names>D</given-names> </name><name name-style="western"><surname>Poxleitner</surname><given-names>P</given-names> </name><name name-style="western"><surname>Aljohani</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Evaluation of a context-aware chatbot using retrieval-augmented generation for answering clinical questions on medication-related osteonecrosis of the jaw</article-title><source>J Craniomaxillofac Surg</source><year>2025</year><month>04</month><volume>53</volume><issue>4</issue><fpage>355</fpage><lpage>360</lpage><pub-id pub-id-type="doi">10.1016/j.jcms.2024.12.009</pub-id><pub-id pub-id-type="medline">39799075</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Yu</surname><given-names>D</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>S</given-names> </name><etal/></person-group><article-title>YpathRAG: a retrieval-augmented generation framework and benchmark for pathology</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 7, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2510.08603</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ye</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Enhancement of the performance of large language models in diabetes education through retrieval-augmented generation: comparative study</article-title><source>J Med Internet Res</source><year>2024</year><month>11</month><day>8</day><volume>26</volume><fpage>e58041</fpage><pub-id pub-id-type="doi">10.2196/58041</pub-id><pub-id pub-id-type="medline">39046096</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>D</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>W</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Cao</surname><given-names>L</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>K</given-names> </name></person-group><article-title>CliCARE: grounding large language models in clinical guidelines for decision support over longitudinal cancer electronic health records</article-title><source>AAAI</source><year>2026</year><volume>40</volume><issue>37</issue><fpage>31554</fpage><lpage>31562</lpage><pub-id pub-id-type="doi">10.1609/aaai.v40i37.40421</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Busch</surname><given-names>F</given-names> </name><name name-style="western"><surname>Kaibel</surname><given-names>L</given-names> </name><name name-style="western"><surname>Nguyen</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Evaluation of a retrieval-augmented generation-powered chatbot for Pre-CT informed consent: a prospective comparative study</article-title><source>J Imaging Inform Med</source><year>2025</year><month>12</month><volume>38</volume><issue>6</issue><fpage>4312</fpage><lpage>4323</lpage><pub-id pub-id-type="doi">10.1007/s10278-025-01483-w</pub-id><pub-id pub-id-type="medline">40119020</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>H</given-names> </name><name name-style="western"><surname>Sohn</surname><given-names>J</given-names> </name><name name-style="western"><surname>Gilson</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Rethinking retrieval-augmented generation for medicine: a large-scale, systematic expert evaluation and practical insights</article-title><source>arXiv</source><comment>Preprint posted online on  Nov 10, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2511.06738</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Sohn</surname><given-names>J</given-names> </name><name name-style="western"><surname>Park</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yoon</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Rationale-guided retrieval augmented generation for medical question answering</article-title><conf-name>Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics</conf-name><conf-date>Apr 29 to May 4, 2025</conf-date><pub-id pub-id-type="doi">10.18653/v1/2025.naacl-long.635</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Qi</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Medical graph RAG: evidence-based medical large language model via graph retrieval-augmented generation</article-title><conf-name>Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)</conf-name><conf-date>Jul 27 to Aug 1, 2025</conf-date><conf-loc>Vienna, Austria</conf-loc><fpage>28443</fpage><lpage>28467</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.acl-long.1381</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Ou</surname><given-names>J</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>T</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>P</given-names> </name><name name-style="western"><surname>Ying</surname><given-names>R</given-names> </name></person-group><article-title>Experience retrieval-augmentation with electronic health records enables accurate discharge QA</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 23, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2503.17933</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Soman</surname><given-names>K</given-names> </name><name name-style="western"><surname>Rose</surname><given-names>PW</given-names> </name><name name-style="western"><surname>Morris</surname><given-names>JH</given-names> </name><etal/></person-group><article-title>Biomedical knowledge graph-optimized prompt generation for large language models</article-title><source>Bioinformatics</source><year>2024</year><month>09</month><day>2</day><volume>40</volume><issue>9</issue><fpage>btae560</fpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btae560</pub-id><pub-id pub-id-type="medline">39288310</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wo&#x0142;k</surname><given-names>K</given-names> </name></person-group><article-title>Evaluating retrieval-augmented generation variants for clinical decision support: hallucination mitigation and secure on-premises deployment</article-title><source>Electronics (Basel)</source><year>2025</year><volume>14</volume><issue>21</issue><fpage>4227</fpage><pub-id pub-id-type="doi">10.3390/electronics14214227</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Park</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yoon</surname><given-names>B</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>S</given-names> </name><name name-style="western"><surname>Choi</surname><given-names>K</given-names> </name></person-group><article-title>RA-RRG: multimodal retrieval-augmented radiology report generation with key phrase extraction</article-title><conf-name>Findings of the Association for Computational Linguistics</conf-name><conf-date>Jul 2-7, 2026</conf-date><pub-id pub-id-type="doi">10.18653/v1/2026.findings-acl.247</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Garza</surname><given-names>L</given-names> </name><name name-style="western"><surname>Kotal</surname><given-names>A</given-names> </name><name name-style="western"><surname>Grasso</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Umucu</surname><given-names>E</given-names> </name></person-group><article-title>Retrieval-augmented framework for LLM-based clinical decision support</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 1, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2510.01363</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Stuhlmann</surname><given-names>L</given-names> </name><name name-style="western"><surname>Saxer</surname><given-names>MA</given-names> </name><name name-style="western"><surname>F&#x00FC;rst</surname><given-names>J</given-names> </name></person-group><article-title>Efficient and reproducible biomedical question answering using retrieval augmented generation</article-title><conf-name>2025 IEEE Swiss Conference on Data Science (SDS)</conf-name><conf-date>Jun 26-27, 2025</conf-date><pub-id pub-id-type="doi">10.1109/SDS66131.2025.00029</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xiong</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zeng</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Luo</surname><given-names>W</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>R</given-names> </name></person-group><article-title>Nursing retrieval-augmented generation: retrieval augmented generation for nursing question answering with large language models</article-title><source>Int J Nurs Sci</source><year>2025</year><month>11</month><volume>12</volume><issue>6</issue><fpage>516</fpage><lpage>523</lpage><pub-id pub-id-type="doi">10.1016/j.ijnss.2025.10.005</pub-id><pub-id pub-id-type="medline">41367595</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Sun</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Han</surname><given-names>W</given-names> </name><name name-style="western"><surname>Xiong</surname><given-names>C</given-names> </name></person-group><article-title>Fact-aware multimodal retrieval augmentation for accurate medical radiology report generation</article-title><conf-name>Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics</conf-name><conf-date>Apr 29 to May 4, 2025</conf-date><pub-id pub-id-type="doi">10.18653/v1/2025.naacl-long.28</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Sesen</surname><given-names>MB</given-names> </name><name name-style="western"><surname>Au Yeung</surname><given-names>J</given-names> </name><name name-style="western"><surname>Asgari</surname><given-names>E</given-names> </name></person-group><article-title>Development and validation of Retrieval Augmented Generation (RAG) and GraphRAG for complex clinical cases</article-title><source>medRxiv</source><comment>Preprint posted online on  Nov 27, 2025</comment><pub-id pub-id-type="doi">10.1101/2025.11.25.25341010</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moureau</surname><given-names>MK</given-names> </name><name name-style="western"><surname>Davis</surname><given-names>B</given-names> </name><name name-style="western"><surname>Hong</surname><given-names>CX</given-names> </name></person-group><article-title>Development and evaluation of an augmented artificial intelligence model for urogynecology queries</article-title><source>Int Urogynecol J</source><year>2026</year><month>05</month><volume>37</volume><issue>5</issue><fpage>1423</fpage><lpage>1428</lpage><pub-id pub-id-type="doi">10.1007/s00192-025-06446-x</pub-id><pub-id pub-id-type="medline">41364238</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Lewis</surname><given-names>M</given-names> </name><name name-style="western"><surname>Thio</surname><given-names>S</given-names> </name><name name-style="western"><surname>Roberts</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Grounding large language models in clinical evidence: a retrieval-augmented generation system for querying UK NICE clinical guidelines</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 3, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2510.02967</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Welsh</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lopez-Rippe</surname><given-names>J</given-names> </name><name name-style="western"><surname>Alkhulaifat</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Custom-tailored radiology research via retrieval-augmented generation: a secure institutionally deployed large language model system</article-title><source>Inventions</source><year>2025</year><volume>10</volume><issue>4</issue><fpage>55</fpage><pub-id pub-id-type="doi">10.3390/inventions10040055</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alkhalaf</surname><given-names>M</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>P</given-names> </name><name name-style="western"><surname>Yin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Deng</surname><given-names>C</given-names> </name></person-group><article-title>Applying generative AI with retrieval augmented generation to summarize and extract key clinical information from electronic health records</article-title><source>J Biomed Inform</source><year>2024</year><month>08</month><volume>156</volume><fpage>104662</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2024.104662</pub-id><pub-id pub-id-type="medline">38880236</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Rezaei</surname><given-names>MR</given-names> </name><name name-style="western"><surname>Fard</surname><given-names>RS</given-names> </name><name name-style="western"><surname>Parker</surname><given-names>JL</given-names> </name><name name-style="western"><surname>Krishnan</surname><given-names>RG</given-names> </name><name name-style="western"><surname>Lankarany</surname><given-names>M</given-names> </name></person-group><article-title>Agentic Medical Knowledge Graphs Enhance Medical Question Answering: Bridging the Gap Between LLMs and Evolving Medical Knowledge</article-title><access-date>2026-07-06</access-date><conf-name>Findings of the Association for Computational Linguistics</conf-name><conf-date>Nov 4-9, 2025</conf-date><conf-loc>Suzhou, China</conf-loc><fpage>12682</fpage><lpage>12701</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2025.findings-emnlp.679/">https://aclanthology.org/2025.findings-emnlp.679/</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/2025.findings-emnlp.679</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Son</surname><given-names>N</given-names> </name><name name-style="western"><surname>Kang</surname><given-names>I</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>I</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>K</given-names> </name><name name-style="western"><surname>Nam</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>D</given-names> </name></person-group><article-title>Development and evaluation of a retrieval-augmented generation-based electronic medical record chatbot system</article-title><source>Healthc Inform Res</source><year>2025</year><month>07</month><volume>31</volume><issue>3</issue><fpage>218</fpage><lpage>225</lpage><pub-id pub-id-type="doi">10.4258/hir.2025.31.3.218</pub-id><pub-id pub-id-type="medline">40840929</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abdullayev</surname><given-names>N</given-names> </name><name name-style="western"><surname>Kottlors</surname><given-names>J</given-names> </name><name name-style="western"><surname>Habibov</surname><given-names>H</given-names> </name><etal/></person-group><article-title>European guideline informed RAG-based GPT-4 decision support tool in tumor board meetings for breast cancer treatment</article-title><source>Eur J Surg Oncol</source><year>2025</year><month>11</month><volume>51</volume><issue>11</issue><fpage>110384</fpage><pub-id pub-id-type="doi">10.1016/j.ejso.2025.110384</pub-id><pub-id pub-id-type="medline">40845749</pub-id></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Valan</surname><given-names>P</given-names> </name><name name-style="western"><surname>Venugopal</surname><given-names>P</given-names> </name></person-group><article-title>Evaluating a retrieval-augmented pregnancy chatbot: a comprehensibility-accuracy-readability study of the DIAN AI assistant</article-title><source>Front Artif Intell</source><year>2025</year><volume>8</volume><fpage>1640994</fpage><pub-id pub-id-type="doi">10.3389/frai.2025.1640994</pub-id><pub-id pub-id-type="medline">41058908</pub-id></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Xia</surname><given-names>P</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>K</given-names> </name><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><etal/></person-group><article-title>RULE: reliable multimodal RAG for factuality in medical vision language models</article-title><conf-name>Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Nov 12-16, 2024</conf-date><pub-id pub-id-type="doi">10.18653/v1/2024.emnlp-main.62</pub-id></nlm-citation></ref><ref id="ref64"><label>64</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>DiGiacomo</surname><given-names>P</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Fang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Leng</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Brode</surname><given-names>WM</given-names> </name><name name-style="western"><surname>Ding</surname><given-names>Y</given-names> </name></person-group><article-title>Demo: guide-RAG: evidence-driven corpus curation for retrieval-augmented generation in long COVID</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 17, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2510.15782</pub-id></nlm-citation></ref><ref id="ref65"><label>65</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>R</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zheng</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>C</given-names> </name></person-group><article-title>Enhancing treatment decision-making for low back pain: a novel framework integrating large language models with retrieval-augmented generation technology</article-title><source>Front Med (Lausanne)</source><year>2025</year><volume>12</volume><fpage>1599241</fpage><pub-id pub-id-type="doi">10.3389/fmed.2025.1599241</pub-id><pub-id pub-id-type="medline">40438365</pub-id></nlm-citation></ref><ref id="ref66"><label>66</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wind</surname><given-names>S</given-names> </name><name name-style="western"><surname>Sopa</surname><given-names>J</given-names> </name><name name-style="western"><surname>Truhn</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Multi-step retrieval and reasoning improves radiology question answering with large language models</article-title><source>NPJ Digit Med</source><year>2025</year><month>12</month><day>22</day><volume>8</volume><issue>1</issue><fpage>790</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-02250-5</pub-id><pub-id pub-id-type="medline">41429891</pub-id></nlm-citation></ref><ref id="ref67"><label>67</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kuo</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Tai</surname><given-names>SK</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>HY</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>RC</given-names> </name></person-group><article-title>Automated clinical trial data analysis and report generation by integrating retrieval-augmented generation (RAG) and large language model (LLM) technologies</article-title><source>AI</source><year>2025</year><volume>6</volume><issue>8</issue><fpage>188</fpage><pub-id pub-id-type="doi">10.3390/ai6080188</pub-id></nlm-citation></ref><ref id="ref68"><label>68</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Liang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>W</given-names> </name><name name-style="western"><surname>He</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>D</given-names> </name></person-group><article-title>RGAR: recurrence generation-augmented retrieval for factual-aware medical question answering</article-title><conf-name>Findings of the Association for Computational Linguistics</conf-name><conf-date>Nov 4-9, 2025</conf-date><pub-id pub-id-type="doi">10.18653/v1/2025.findings-emnlp.214</pub-id></nlm-citation></ref><ref id="ref69"><label>69</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>S</given-names> </name><name name-style="western"><surname>An</surname><given-names>LCI</given-names> </name><name name-style="western"><surname>Mihalcea</surname><given-names>R</given-names> </name></person-group><article-title>Patient-centered RAG for oncology visit aid following the Ottawa Decision Guide</article-title><source>arXiv</source><comment>Preprint posted online on  Jul 5, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2507.04026</pub-id></nlm-citation></ref><ref id="ref70"><label>70</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Myers</surname><given-names>S</given-names> </name><name name-style="western"><surname>Dligach</surname><given-names>D</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>TA</given-names> </name><etal/></person-group><article-title>Evaluating retrieval-augmented generation vs. long-context input for clinical reasoning over EHRs</article-title><source>arXiv</source><year>2025</year><month>08</month><day>20</day><pub-id pub-id-type="doi">10.48550/arXiv.2508.14817</pub-id></nlm-citation></ref><ref id="ref71"><label>71</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Hassan</surname><given-names>T</given-names> </name><name name-style="western"><surname>Karim</surname><given-names>MF</given-names> </name><name name-style="western"><surname>Jeelani</surname><given-names>H</given-names> </name><name name-style="western"><surname>Behnam</surname><given-names>E</given-names> </name><name name-style="western"><surname>Green</surname><given-names>R</given-names> </name><name name-style="western"><surname>Syed</surname><given-names>FJ</given-names> </name></person-group><article-title>Optimizing medical question-answering systems: a comparative study of fine-tuned and zero-shot large language models with RAG framework</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 5, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2512.05863</pub-id></nlm-citation></ref><ref id="ref72"><label>72</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Sekar</surname><given-names>T</given-names> </name><name name-style="western"><surname>Kushal</surname></name><name name-style="western"><surname>Shankar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Mohammed</surname><given-names>S</given-names> </name><name name-style="western"><surname>Fiaidhi</surname><given-names>J</given-names> </name></person-group><article-title>Investigations on using evidence-based GraphRag pipeline using LLM tailored for USMLE style questions</article-title><source>medRxiv</source><comment>Preprint posted online on  May 5, 2025</comment><pub-id pub-id-type="doi">10.1101/2025.05.03.25325604</pub-id></nlm-citation></ref><ref id="ref73"><label>73</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Parameswaran</surname><given-names>V</given-names> </name><name name-style="western"><surname>Bernard</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bernard</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Evaluating large language models and retrieval-augmented generation enhancement for delivering guideline-adherent nutrition information for cardiovascular disease prevention: cross-sectional study</article-title><source>J Med Internet Res</source><year>2025</year><month>10</month><day>7</day><volume>27</volume><fpage>e78625</fpage><pub-id pub-id-type="doi">10.2196/78625</pub-id><pub-id pub-id-type="medline">41057043</pub-id></nlm-citation></ref><ref id="ref74"><label>74</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Guo</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Patho-AgenticRAG: towards multimodal agentic retrieval-augmented generation for pathology VLMs via reinforcement learning</article-title><source>AAAI</source><year>2026</year><volume>40</volume><issue>35</issue><fpage>29921</fpage><lpage>29929</lpage><pub-id pub-id-type="doi">10.1609/aaai.v40i35.40239</pub-id></nlm-citation></ref><ref id="ref75"><label>75</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Development and evaluation of a retrieval-augmented large language model framework for enhancing endodontic education</article-title><source>Int J Med Inform</source><year>2025</year><month>11</month><volume>203</volume><fpage>106006</fpage><pub-id pub-id-type="doi">10.1016/j.ijmedinf.2025.106006</pub-id><pub-id pub-id-type="medline">40479778</pub-id></nlm-citation></ref><ref id="ref76"><label>76</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Dong</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>W</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Talk before you retrieve: agent-led discussions for better RAG in medical QA</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 30, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2504.21252</pub-id></nlm-citation></ref><ref id="ref77"><label>77</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zhao</surname><given-names>X</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>SY</given-names> </name><name name-style="western"><surname>Miao</surname><given-names>C</given-names> </name></person-group><article-title>MedRAG: enhancing retrieval-augmented generation with knowledge graph-elicited reasoning for healthcare copilot</article-title><conf-name>WWW &#x2019;25</conf-name><conf-date>Apr 28 to May 2, 2025</conf-date><pub-id pub-id-type="doi">10.1145/3696410.3714782</pub-id></nlm-citation></ref><ref id="ref78"><label>78</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ge</surname><given-names>X</given-names> </name><name name-style="western"><surname>Murtaza</surname><given-names>S</given-names> </name><name name-style="western"><surname>Cortez</surname><given-names>A</given-names> </name><name name-style="western"><surname>Alemzadeh</surname><given-names>H</given-names> </name></person-group><article-title>Expert-guided prompting and retrieval-augmented generation for emergency medical service question answering</article-title><source>AAAI</source><year>2025</year><volume>40</volume><issue>36</issue><fpage>30798</fpage><lpage>30806</lpage><pub-id pub-id-type="doi">10.1609/aaai.v40i36.40337</pub-id></nlm-citation></ref><ref id="ref79"><label>79</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Mali</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zeng</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Heo</surname><given-names>K</given-names> </name><etal/></person-group><article-title>A chatbot for the management of bipolar disorder: using retrieval-augmented generation with an open-weight large language model to answer clinical questions based on the CANMAT and ISBD 2018 guidelines for bipolar disorder</article-title><source>medRxiv</source><comment>Preprint posted online on  Jan 7, 2026</comment><pub-id pub-id-type="doi">10.64898/2025.11.30.25341311</pub-id></nlm-citation></ref><ref id="ref80"><label>80</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Yu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Li</surname><given-names>L</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name></person-group><article-title>Augmenting large language models and retrieval-augmented generation with an evidence-based medicine-enabled agent system</article-title><source>medRxiv</source><comment>Preprint posted online on  Oct 20, 2025</comment><pub-id pub-id-type="doi">10.1101/2025.10.17.25338266</pub-id></nlm-citation></ref><ref id="ref81"><label>81</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ke</surname><given-names>YH</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>L</given-names> </name><name name-style="western"><surname>Elangovan</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Retrieval augmented generation for 10 large language models and its generalizability in assessing medical fitness</article-title><source>NPJ Digit Med</source><year>2025</year><month>04</month><day>5</day><volume>8</volume><issue>1</issue><fpage>187</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01519-z</pub-id><pub-id pub-id-type="medline">40185842</pub-id></nlm-citation></ref><ref id="ref82"><label>82</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Ji</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name></person-group><article-title>Bias evaluation and mitigation in retrieval-augmented medical question-answering systems</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 19, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2503.15454</pub-id></nlm-citation></ref><ref id="ref83"><label>83</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Xue</surname><given-names>K</given-names> </name><name name-style="western"><surname>Fan</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Tool calling: enhancing medication consultation via retrieval-augmented large language models</article-title><source>arXiv</source><year>2024</year><month>04</month><day>27</day><pub-id pub-id-type="doi">10.48550/arXiv.2404.17897</pub-id></nlm-citation></ref><ref id="ref84"><label>84</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>J</given-names> </name><name name-style="western"><surname>Danek</surname><given-names>B</given-names> </name><etal/></person-group><article-title>InformGen: an AI copilot for accurate and compliant clinical research consent document generation</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 1, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2504.00934</pub-id></nlm-citation></ref><ref id="ref85"><label>85</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Khatibi</surname><given-names>E</given-names> </name><name name-style="western"><surname>Rahmani</surname><given-names>AM</given-names> </name></person-group><article-title>MedCoT-RAG: causal chain-of-thought RAG for medical question answering</article-title><conf-name>2025 IEEE 21st International Conference on Body Sensor Networks (BSN)</conf-name><conf-date>Nov 3-5, 2025</conf-date><pub-id pub-id-type="doi">10.1109/BSN66969.2025.11337389</pub-id></nlm-citation></ref><ref id="ref86"><label>86</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>R</given-names> </name><name name-style="western"><surname>Hong</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>F</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>H</given-names> </name></person-group><article-title>Evaluation of the integration of retrieval-augmented generation in large language model for breast cancer nursing care responses</article-title><source>Sci Rep</source><year>2024</year><month>12</month><day>28</day><volume>14</volume><issue>1</issue><fpage>30794</fpage><pub-id pub-id-type="doi">10.1038/s41598-024-81052-3</pub-id><pub-id pub-id-type="medline">39730573</pub-id></nlm-citation></ref><ref id="ref87"><label>87</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Luo</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Pang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bi</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Development and evaluation of a retrieval-augmented large language model framework for ophthalmology</article-title><source>JAMA Ophthalmol</source><year>2024</year><month>09</month><day>1</day><volume>142</volume><issue>9</issue><fpage>798</fpage><lpage>805</lpage><pub-id pub-id-type="doi">10.1001/jamaophthalmol.2024.2513</pub-id><pub-id pub-id-type="medline">39023885</pub-id></nlm-citation></ref><ref id="ref88"><label>88</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>G</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>Q</given-names> </name><etal/></person-group><article-title>Leveraging long context in retrieval augmented language models for medical question answering</article-title><source>NPJ Digit Med</source><year>2025</year><month>05</month><day>2</day><volume>8</volume><issue>1</issue><fpage>239</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01651-w</pub-id><pub-id pub-id-type="medline">40316710</pub-id></nlm-citation></ref><ref id="ref89"><label>89</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Low</surname><given-names>YS</given-names> </name><name name-style="western"><surname>Jackson</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Hyde</surname><given-names>RJ</given-names> </name><etal/></person-group><article-title>Answering real-world clinical questions using large language model, retrieval-augmented generation, and agentic systems</article-title><source>Digit Health</source><year>2025</year><volume>11</volume><fpage>20552076251348850</fpage><pub-id pub-id-type="doi">10.1177/20552076251348850</pub-id><pub-id pub-id-type="medline">40510193</pub-id></nlm-citation></ref><ref id="ref90"><label>90</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ge</surname><given-names>J</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>S</given-names> </name><name name-style="western"><surname>Owens</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Development of a liver disease-specific large language model chat interface using retrieval-augmented generation</article-title><source>Hepatology</source><year>2024</year><month>11</month><day>1</day><volume>80</volume><issue>5</issue><fpage>1158</fpage><lpage>1168</lpage><pub-id pub-id-type="doi">10.1097/HEP.0000000000000834</pub-id><pub-id pub-id-type="medline">38451962</pub-id></nlm-citation></ref><ref id="ref91"><label>91</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zakka</surname><given-names>C</given-names> </name><name name-style="western"><surname>Shad</surname><given-names>R</given-names> </name><name name-style="western"><surname>Chaurasia</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Almanac - retrieval-augmented language models for clinical medicine</article-title><source>NEJM AI</source><year>2024</year><month>02</month><volume>1</volume><issue>2</issue><pub-id pub-id-type="doi">10.1056/aioa2300068</pub-id><pub-id pub-id-type="medline">38343631</pub-id></nlm-citation></ref><ref id="ref92"><label>92</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tozuka</surname><given-names>R</given-names> </name><name name-style="western"><surname>Johno</surname><given-names>H</given-names> </name><name name-style="western"><surname>Amakawa</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Application of NotebookLM, a large language model with retrieval-augmented generation, for lung cancer staging</article-title><source>Jpn J Radiol</source><year>2025</year><month>04</month><volume>43</volume><issue>4</issue><fpage>706</fpage><lpage>712</lpage><pub-id pub-id-type="doi">10.1007/s11604-024-01705-1</pub-id><pub-id pub-id-type="medline">39585559</pub-id></nlm-citation></ref><ref id="ref93"><label>93</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hewitt</surname><given-names>KJ</given-names> </name><name name-style="western"><surname>Wiest</surname><given-names>IC</given-names> </name><name name-style="western"><surname>Carrero</surname><given-names>ZI</given-names> </name><etal/></person-group><article-title>Large language models as a diagnostic support tool in neuropathology</article-title><source>J Pathol Clin Res</source><year>2024</year><month>11</month><volume>10</volume><issue>6</issue><fpage>e70009</fpage><pub-id pub-id-type="doi">10.1002/2056-4538.70009</pub-id><pub-id pub-id-type="medline">39505569</pub-id></nlm-citation></ref><ref id="ref94"><label>94</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Ye</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Enhancing large language models for improved accuracy and safety in medical question answering: comparative study</article-title><source>JMIR Med Educ</source><year>2025</year><month>12</month><day>2</day><volume>11</volume><fpage>e70190</fpage><pub-id pub-id-type="doi">10.2196/70190</pub-id><pub-id pub-id-type="medline">41329953</pub-id></nlm-citation></ref><ref id="ref95"><label>95</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Jia</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Enhancing clinical documentation with voice processing and large language models: a study on the LAOS system</article-title><source>NPJ Digit Med</source><year>2025</year><month>11</month><day>28</day><volume>8</volume><issue>1</issue><fpage>798</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-02170-4</pub-id><pub-id pub-id-type="medline">41315671</pub-id></nlm-citation></ref><ref id="ref96"><label>96</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tung</surname><given-names>JYM</given-names> </name><name name-style="western"><surname>Le</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Yao</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Performance of retrieval-augmented generation large language models in guideline-concordant prostate-specific antigen testing: comparative study with junior clinicians</article-title><source>J Med Internet Res</source><year>2025</year><month>11</month><day>19</day><volume>27</volume><fpage>e78393</fpage><pub-id pub-id-type="doi">10.2196/78393</pub-id><pub-id pub-id-type="medline">41259800</pub-id></nlm-citation></ref><ref id="ref97"><label>97</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>X</given-names> </name><etal/></person-group><article-title>A knowledge-enhanced platform (MetaSepsisKnowHub) for retrieval augmented generation-based sepsis heterogeneity and personalized management: development study</article-title><source>J Med Internet Res</source><year>2025</year><month>06</month><day>6</day><volume>27</volume><fpage>e67201</fpage><pub-id pub-id-type="doi">10.2196/67201</pub-id><pub-id pub-id-type="medline">40478618</pub-id></nlm-citation></ref><ref id="ref98"><label>98</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fanelli</surname><given-names>F</given-names> </name><name name-style="western"><surname>Saleh</surname><given-names>M</given-names> </name><name name-style="western"><surname>Santamaria</surname><given-names>P</given-names> </name><name name-style="western"><surname>Zhurakivska</surname><given-names>K</given-names> </name><name name-style="western"><surname>Nibali</surname><given-names>L</given-names> </name><name name-style="western"><surname>Troiano</surname><given-names>G</given-names> </name></person-group><article-title>Development and comparative evaluation of a reinstructed GPT-4o model specialized in periodontology</article-title><source>J Clin Periodontol</source><year>2025</year><month>05</month><volume>52</volume><issue>5</issue><fpage>707</fpage><lpage>716</lpage><pub-id pub-id-type="doi">10.1111/jcpe.14101</pub-id><pub-id pub-id-type="medline">39723544</pub-id></nlm-citation></ref><ref id="ref99"><label>99</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Masanneck</surname><given-names>L</given-names> </name><name name-style="western"><surname>Meuth</surname><given-names>SG</given-names> </name><name name-style="western"><surname>Pawlitzki</surname><given-names>M</given-names> </name></person-group><article-title>Evaluating base and retrieval augmented LLMs with document or online support for evidence based neurology</article-title><source>NPJ Digit Med</source><year>2025</year><month>03</month><day>4</day><volume>8</volume><issue>1</issue><fpage>137</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01536-y</pub-id><pub-id pub-id-type="medline">40038423</pub-id></nlm-citation></ref><ref id="ref100"><label>100</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Zuo</surname><given-names>H</given-names> </name><name name-style="western"><surname>Su</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Dual retrieving and ranking medical large language model with retrieval augmented generation</article-title><source>Sci Rep</source><year>2025</year><month>05</month><day>24</day><volume>15</volume><issue>1</issue><fpage>18062</fpage><pub-id pub-id-type="doi">10.1038/s41598-025-00724-w</pub-id><pub-id pub-id-type="medline">40413225</pub-id></nlm-citation></ref><ref id="ref101"><label>101</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vach</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gliem</surname><given-names>M</given-names> </name><name name-style="western"><surname>Weiss</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Evaluating retrieval augmented generation-enhanced large language models for question answering on German neurovascular guidelines</article-title><source>Clin Neuroradiol</source><year>2026</year><month>03</month><volume>36</volume><issue>1</issue><fpage>119</fpage><lpage>127</lpage><pub-id pub-id-type="doi">10.1007/s00062-025-01562-z</pub-id><pub-id pub-id-type="medline">40892297</pub-id></nlm-citation></ref><ref id="ref102"><label>102</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Masanneck</surname><given-names>L</given-names> </name><name name-style="western"><surname>Epping</surname><given-names>PZ</given-names> </name><name name-style="western"><surname>Meuth</surname><given-names>SG</given-names> </name><name name-style="western"><surname>Pawlitzki</surname><given-names>M</given-names> </name></person-group><article-title>Evaluating web retrieval-assisted large language models with and without whitelisting for evidence-based neurology: comparative study</article-title><source>J Med Internet Res</source><year>2025</year><month>10</month><day>29</day><volume>27</volume><fpage>e79379</fpage><pub-id pub-id-type="doi">10.2196/79379</pub-id><pub-id pub-id-type="medline">41159599</pub-id></nlm-citation></ref><ref id="ref103"><label>103</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fukui</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Kawata</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Kobashi</surname><given-names>K</given-names> </name><name name-style="western"><surname>Nagatani</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Iguchi</surname><given-names>H</given-names> </name></person-group><article-title>Evaluation of a retrieval-augmented generation system using a Japanese institutional nuclear medicine manual and large language model-automated scoring</article-title><source>Radiol Phys Technol</source><year>2025</year><month>09</month><volume>18</volume><issue>3</issue><fpage>861</fpage><lpage>876</lpage><pub-id pub-id-type="doi">10.1007/s12194-025-00941-y</pub-id><pub-id pub-id-type="medline">40683982</pub-id></nlm-citation></ref><ref id="ref104"><label>104</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sha</surname><given-names>H</given-names> </name><name name-style="western"><surname>Gong</surname><given-names>F</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>B</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>R</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>T</given-names> </name></person-group><article-title>Leveraging retrieval-augmented large language models for dietary recommendations with traditional chinese medicine&#x2019;s medicine food homology: algorithm development and validation</article-title><source>JMIR Med Inform</source><year>2025</year><month>08</month><day>21</day><volume>13</volume><fpage>e75279</fpage><pub-id pub-id-type="doi">10.2196/75279</pub-id><pub-id pub-id-type="medline">40840437</pub-id></nlm-citation></ref><ref id="ref105"><label>105</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Owoyemi</surname><given-names>J</given-names> </name><name name-style="western"><surname>Abubakar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Owoyemi</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Open-source retrieval augmented generation framework for retrieving accurate medication insights from formularies for African healthcare workers</article-title><source>medRxiv</source><comment>Preprint posted online on  Feb 21, 2025</comment><pub-id pub-id-type="doi">10.1101/2025.02.20.25322640</pub-id></nlm-citation></ref><ref id="ref106"><label>106</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Tata</surname><given-names>V</given-names> </name><name name-style="western"><surname>Bouchamaoui</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Bhaskara</surname><given-names>NV</given-names> </name></person-group><article-title>OrthoGraphRAG: enhancing clinical decision making with multi-level knowledge graphs</article-title><source>ICML 2025 GenBio Workshop Poster (OpenReview)</source><year>2025</year><month>06</month><access-date>2026-07-06</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://openreview.net/pdf?id=ht8uX6Pj0d">https://openreview.net/pdf?id=ht8uX6Pj0d</ext-link></comment></nlm-citation></ref><ref id="ref107"><label>107</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tayebi Arasteh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lotfinia</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bressem</surname><given-names>K</given-names> </name><etal/></person-group><article-title>RadioRAG: online retrieval-augmented generation for radiology question answering</article-title><source>Radiol Artif Intell</source><year>2025</year><month>07</month><volume>7</volume><issue>4</issue><fpage>e240476</fpage><pub-id pub-id-type="doi">10.1148/ryai.240476</pub-id><pub-id pub-id-type="medline">40530957</pub-id></nlm-citation></ref><ref id="ref108"><label>108</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Das</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ge</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Guo</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Two-layer retrieval-augmented generation framework for low-resource medical question answering using Reddit data: proof-of-concept study</article-title><source>J Med Internet Res</source><year>2025</year><month>01</month><day>6</day><volume>27</volume><fpage>e66220</fpage><pub-id pub-id-type="doi">10.2196/66220</pub-id><pub-id pub-id-type="medline">39761554</pub-id></nlm-citation></ref><ref id="ref109"><label>109</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ji</surname><given-names>M</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>An</surname><given-names>R</given-names> </name></person-group><article-title>Use of retrieval-augmented large language model for COVID-19 fact-checking: development and usability study</article-title><source>J Med Internet Res</source><year>2025</year><month>04</month><day>30</day><volume>27</volume><fpage>e66098</fpage><pub-id pub-id-type="doi">10.2196/66098</pub-id><pub-id pub-id-type="medline">40306628</pub-id></nlm-citation></ref><ref id="ref110"><label>110</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Song</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>MG</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>HW</given-names> </name><name name-style="western"><surname>Choe</surname><given-names>EK</given-names> </name><name name-style="western"><surname>Chai</surname><given-names>YJ</given-names> </name></person-group><article-title>Thyro-GenAI: a chatbot using retrieval-augmented generative models for personalized thyroid disease management</article-title><source>J Clin Med</source><year>2025</year><month>04</month><day>3</day><volume>14</volume><issue>7</issue><fpage>2450</fpage><pub-id pub-id-type="doi">10.3390/jcm14072450</pub-id><pub-id pub-id-type="medline">40217905</pub-id></nlm-citation></ref><ref id="ref111"><label>111</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aguzzi</surname><given-names>G</given-names> </name><name name-style="western"><surname>Magnini</surname><given-names>M</given-names> </name><name name-style="western"><surname>Farahmand</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ferretti</surname><given-names>S</given-names> </name><name name-style="western"><surname>Pengo</surname><given-names>MF</given-names> </name><name name-style="western"><surname>Montagna</surname><given-names>S</given-names> </name></person-group><article-title>RAG-enhanced open SLMs for hypertension management chatbots</article-title><source>J Med Syst</source><year>2025</year><month>11</month><day>13</day><volume>49</volume><issue>1</issue><fpage>159</fpage><pub-id pub-id-type="doi">10.1007/s10916-025-02297-7</pub-id><pub-id pub-id-type="medline">41231304</pub-id></nlm-citation></ref><ref id="ref112"><label>112</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhou</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Duan</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>GastroBot: a Chinese gastrointestinal disease chatbot based on the retrieval-augmented generation</article-title><source>Front Med (Lausanne)</source><year>2024</year><volume>11</volume><fpage>1392555</fpage><pub-id pub-id-type="doi">10.3389/fmed.2024.1392555</pub-id><pub-id pub-id-type="medline">38841582</pub-id></nlm-citation></ref><ref id="ref113"><label>113</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Aminan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Darnell</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Delsoz</surname><given-names>M</given-names> </name><etal/></person-group><article-title>GlaucoRAG: a retrieval-augmented large language model for expert-level glaucoma assessment</article-title><source>medRxiv</source><comment>Preprint posted online on  Jul 7, 2025</comment><pub-id pub-id-type="doi">10.1101/2025.07.03.25330805</pub-id><pub-id pub-id-type="medline">40672509</pub-id></nlm-citation></ref><ref id="ref114"><label>114</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Hsu</surname><given-names>HL</given-names> </name><name name-style="western"><surname>Dao</surname><given-names>CT</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name><etal/></person-group><article-title>MedPlan: a two-stage RAG-based system for personalized medical plan generation</article-title><conf-name>Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 6: Industry Track)</conf-name><conf-date>Jul 28-30, 2025</conf-date><conf-loc>Vienna, Austria</conf-loc><fpage>1072</fpage><lpage>1082</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.acl-industry.76</pub-id></nlm-citation></ref><ref id="ref115"><label>115</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kresevic</surname><given-names>S</given-names> </name><name name-style="western"><surname>Giuffr&#x00E8;</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ajcevic</surname><given-names>M</given-names> </name><name name-style="western"><surname>Accardo</surname><given-names>A</given-names> </name><name name-style="western"><surname>Croc&#x00E8;</surname><given-names>LS</given-names> </name><name name-style="western"><surname>Shung</surname><given-names>DL</given-names> </name></person-group><article-title>Optimization of hepatological clinical guidelines interpretation by large language models: a retrieval augmented generation-based framework</article-title><source>NPJ Digit Med</source><year>2024</year><month>04</month><day>23</day><volume>7</volume><issue>1</issue><fpage>102</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01091-y</pub-id><pub-id pub-id-type="medline">38654102</pub-id></nlm-citation></ref><ref id="ref116"><label>116</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Kang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yun</surname><given-names>TR</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>CE</given-names> </name></person-group><article-title>Prompt-RAG: pioneering vector embedding-free retrieval-augmented generation in niche domains, exemplified by korean medicine</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 20, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2401.11246</pub-id></nlm-citation></ref><ref id="ref117"><label>117</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>AlSammarraie</surname><given-names>A</given-names> </name><name name-style="western"><surname>Al-Saifi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kamhia</surname><given-names>H</given-names> </name><name name-style="western"><surname>Aboagla</surname><given-names>M</given-names> </name><name name-style="western"><surname>Househ</surname><given-names>M</given-names> </name></person-group><article-title>Development and evaluation of an agentic LLM based RAG framework for evidence-based patient education</article-title><source>BMJ Health Care Inform</source><year>2025</year><month>07</month><day>25</day><volume>32</volume><issue>1</issue><fpage>e101570</fpage><pub-id pub-id-type="doi">10.1136/bmjhci-2025-101570</pub-id><pub-id pub-id-type="medline">40713064</pub-id></nlm-citation></ref><ref id="ref118"><label>118</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Nandy</surname><given-names>G</given-names> </name><name name-style="western"><surname>Gupta</surname><given-names>S</given-names> </name><name name-style="western"><surname>Mohammad Afzali</surname><given-names>F</given-names> </name><name name-style="western"><surname>Peeples</surname><given-names>E</given-names> </name><name name-style="western"><surname>Pilon</surname><given-names>B</given-names> </name><name name-style="western"><surname>Tsai</surname><given-names>CH</given-names> </name></person-group><article-title>Balancing health information-seeking through retrieval-augmented generation-based LLM chatbot</article-title><conf-name>UMAP &#x2019;25</conf-name><conf-date>Jun 16-19, 2025</conf-date><pub-id pub-id-type="doi">10.1145/3708319.3733709</pub-id></nlm-citation></ref><ref id="ref119"><label>119</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hetz</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Carl</surname><given-names>N</given-names> </name><name name-style="western"><surname>Haggenm&#x00FC;ller</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Superhuman performance on urology board questions using an explainable language model enhanced with European Association of Urology guidelines</article-title><source>ESMO Real World Data Digit Oncol</source><year>2024</year><month>12</month><volume>6</volume><fpage>100078</fpage><pub-id pub-id-type="doi">10.1016/j.esmorw.2024.100078</pub-id><pub-id pub-id-type="medline">41646097</pub-id></nlm-citation></ref><ref id="ref120"><label>120</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ong</surname><given-names>CS</given-names> </name><name name-style="western"><surname>Obey</surname><given-names>NT</given-names> </name><name name-style="western"><surname>Zheng</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Cohan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Schneider</surname><given-names>EB</given-names> </name></person-group><article-title>SurgeryLLM: a retrieval-augmented generation large language model framework for surgical decision support and workflow enhancement</article-title><source>NPJ Digit Med</source><year>2024</year><month>12</month><day>18</day><volume>7</volume><issue>1</issue><fpage>364</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01391-3</pub-id><pub-id pub-id-type="medline">39695316</pub-id></nlm-citation></ref><ref id="ref121"><label>121</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Carl</surname><given-names>N</given-names> </name><name name-style="western"><surname>Hetz</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Wies</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Enhancing clinicians&#x2019; trust in large language models via transparent source attribution: a randomized controlled evaluation in uro-oncology</article-title><source>Eur J Cancer</source><year>2026</year><month>01</month><day>17</day><volume>233</volume><fpage>116168</fpage><pub-id pub-id-type="doi">10.1016/j.ejca.2025.116168</pub-id><pub-id pub-id-type="medline">41401634</pub-id></nlm-citation></ref><ref id="ref122"><label>122</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Tytler</surname><given-names>K</given-names> </name></person-group><article-title>Adoption, usability and perceived clinical value of a UK AI clinical reference platform: a mixed-methods formative evaluation of real-world usage and a 1,223-respondent user survey</article-title><source>arXiv</source><comment>Preprint posted online on  Sep 25, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2509.21188</pub-id></nlm-citation></ref><ref id="ref123"><label>123</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>W</given-names> </name><etal/></person-group><article-title>ChatMyopia: an AI agent for myopia-related consultation in primary eye care settings</article-title><source>iScience</source><year>2025</year><month>11</month><volume>28</volume><issue>11</issue><fpage>113768</fpage><pub-id pub-id-type="doi">10.1016/j.isci.2025.113768</pub-id><pub-id pub-id-type="medline">41244561</pub-id></nlm-citation></ref><ref id="ref124"><label>124</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>S</given-names> </name></person-group><article-title>MedBioRAG: semantic search and retrieval-augmented generation with large language models for medical and biological QA</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 10, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2512.10996</pub-id></nlm-citation></ref><ref id="ref125"><label>125</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Hasan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hossain</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Sayem</surname><given-names>FH</given-names> </name><etal/></person-group><article-title>CLIN-LLM: a safety-constrained hybrid framework for clinical diagnosis and treatment generation</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 26, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2510.22609</pub-id></nlm-citation></ref><ref id="ref126"><label>126</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Long</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>G</given-names> </name><etal/></person-group><article-title>KidneyTalk-open: no-code deployment of a private large language model with medical documentation-enhanced knowledge database for kidney disease</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 6, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2503.04153</pub-id></nlm-citation></ref><ref id="ref127"><label>127</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Song</surname><given-names>J</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>W</given-names> </name></person-group><article-title>From evidence-based medicine to knowledge graph: retrieval-augmented generation for sports rehabilitation and a domain benchmark</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 1, 2026</comment><pub-id pub-id-type="doi">10.48550/arXiv.2601.00216</pub-id></nlm-citation></ref><ref id="ref128"><label>128</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Ryan</surname><given-names>J</given-names> </name><name name-style="western"><surname>Gumilang</surname><given-names>AI</given-names> </name><name name-style="western"><surname>Wiliam</surname><given-names>R</given-names> </name><name name-style="western"><surname>Suhartono</surname><given-names>D</given-names> </name></person-group><article-title>Self-MedRAG: a self-reflective hybrid retrieval-augmented generation framework for reliable medical question answering</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 8, 2026</comment><pub-id pub-id-type="doi">10.48550/arXiv.2601.04531</pub-id></nlm-citation></ref><ref id="ref129"><label>129</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Xue</surname><given-names>H</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>T</given-names> </name></person-group><article-title>Making medical vision-language models think causally across modalities with retrieval-augmented cross-modal reasoning</article-title><conf-name>ICASSP 2026 - 2026 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</conf-name><conf-date>May 3-8, 2026</conf-date><pub-id pub-id-type="doi">10.1109/ICASSP55912.2026.11462104</pub-id></nlm-citation></ref><ref id="ref130"><label>130</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Lorenzo</surname><given-names>L</given-names> </name><name name-style="western"><surname>Montana-Mendez</surname><given-names>M</given-names> </name><name name-style="western"><surname>Figueiras</surname><given-names>S</given-names> </name><name name-style="western"><surname>Boubeta</surname><given-names>B</given-names> </name><name name-style="western"><surname>Bernardo-Castineira</surname><given-names>C</given-names> </name></person-group><article-title>Evaluation of oncotimia: an LLM based system for supporting tumour boards</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 27, 2026</comment><pub-id pub-id-type="doi">10.48550/arXiv.2601.19899</pub-id></nlm-citation></ref><ref id="ref131"><label>131</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>MedCoRAG: interpretable hepatology diagnosis via hybrid evidence retrieval and multispecialty consensus</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 5, 2026</comment><pub-id pub-id-type="doi">10.48550/arXiv.2603.05129</pub-id></nlm-citation></ref><ref id="ref132"><label>132</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Samanta</surname><given-names>HS</given-names> </name></person-group><article-title>Grounded multimodal retrieval-augmented drafting of radiology impressions using case-based similarity search</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 18, 2026</comment><pub-id pub-id-type="doi">10.48550/arXiv.2603.17765</pub-id></nlm-citation></ref><ref id="ref133"><label>133</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Chan</surname><given-names>RWC</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>YN</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>H</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Fan</surname><given-names>W</given-names> </name></person-group><article-title>PriHA: a RAG-enhanced LLM framework for primary healthcare assistant in Hong Kong</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 10, 2026</comment><pub-id pub-id-type="doi">10.48550/arXiv.2604.14215</pub-id></nlm-citation></ref><ref id="ref134"><label>134</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>X</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>B</given-names> </name><name name-style="western"><surname>Le</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Iterative multimodal retrieval-augmented generation for medical question answering</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 30, 2026</comment><pub-id pub-id-type="doi">10.48550/arXiv.2604.27724</pub-id></nlm-citation></ref><ref id="ref135"><label>135</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Khosa</surname><given-names>T</given-names> </name><name name-style="western"><surname>Daramola</surname><given-names>O</given-names> </name></person-group><article-title>Development and preliminary evaluation of a domain-specific large language model for tuberculosis care in South Africa</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 28, 2026</comment><pub-id pub-id-type="doi">10.48550/arXiv.2604.19776</pub-id></nlm-citation></ref><ref id="ref136"><label>136</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Akbar</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Wales-McGrath</surname><given-names>S</given-names> </name><name name-style="western"><surname>Levya</surname><given-names>A</given-names> </name><etal/></person-group><article-title>PathoScribe: transforming pathology data into a living library with a unified LLM-driven framework for semantic retrieval and clinical integration</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 8, 2026</comment><pub-id pub-id-type="doi">10.48550/arXiv.2603.08935</pub-id></nlm-citation></ref><ref id="ref137"><label>137</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>TCM-DiffRAG: personalized syndrome differentiation reasoning method for traditional Chinese medicine based on knowledge graph and chain of thought</article-title><source>Front Med (Lausanne)</source><year>2026</year><month>04</month><day>21</day><volume>13</volume><fpage>1804478</fpage><pub-id pub-id-type="doi">10.3389/fmed.2026.1804478</pub-id></nlm-citation></ref><ref id="ref138"><label>138</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Large language model aided Birt-Hogg-Dube syndrome diagnosis with multimodal retrieval-augmented generation</article-title><source>arXiv</source><comment>Preprint posted online on  Nov 25, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2511.19834</pub-id></nlm-citation></ref><ref id="ref139"><label>139</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Liao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>HeteroRAG: a heterogeneous retrieval-augmented generation framework for medical vision language tasks</article-title><conf-name>Findings of the Association for Computational Linguistics: ACL 2026</conf-name><conf-date>Jul 2-7, 2026</conf-date><pub-id pub-id-type="doi">10.18653/v1/2026.findings-acl.176</pub-id></nlm-citation></ref><ref id="ref140"><label>140</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>W</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Refine medical diagnosis using generation augmented retrieval and clinical practice guidelines</article-title><source>IEEE J Biomed Health Inform</source><year>2025</year><month>12</month><day>9</day><volume>PP</volume><fpage>1</fpage><lpage>14</lpage><pub-id pub-id-type="doi">10.1109/JBHI.2025.3641931</pub-id><pub-id pub-id-type="medline">41364573</pub-id></nlm-citation></ref><ref id="ref141"><label>141</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Ansari</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Khan</surname><given-names>MSA</given-names> </name><name name-style="western"><surname>Revankar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Varma</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mokhade</surname><given-names>AS</given-names> </name></person-group><article-title>Lightweight clinical decision support system using QLoRA-fine-tuned LLMs and retrieval-augmented generation</article-title><source>arXiv</source><comment>Preprint posted online on  May 6, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2505.03406</pub-id></nlm-citation></ref><ref id="ref142"><label>142</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Xia</surname><given-names>P</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>K</given-names> </name><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><etal/></person-group><article-title>MMed-RAG: versatile multimodal RAG system for medical vision language models</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 16, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2410.13085</pub-id></nlm-citation></ref><ref id="ref143"><label>143</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Ning</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Luo</surname><given-names>L</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Pan</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>H</given-names> </name></person-group><article-title>MedTrust-RAG: evidence verification and trust alignment for biomedical question answering</article-title><conf-name>2025 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</conf-name><conf-date>Dec 15-18, 2025</conf-date><pub-id pub-id-type="doi">10.1109/BIBM66473.2025.11356290</pub-id></nlm-citation></ref><ref id="ref144"><label>144</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Kazemzadeh</surname><given-names>H</given-names> </name><name name-style="western"><surname>Dizaji</surname><given-names>KM</given-names> </name><name name-style="western"><surname>Tavakoli</surname><given-names>SR</given-names> </name><etal/></person-group><article-title>DrugRAG: enhancing pharmacy LLM performance through a novel retrieval-augmented generation pipeline</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 16, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2512.14896</pub-id></nlm-citation></ref><ref id="ref145"><label>145</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Li</surname><given-names>F</given-names> </name><name name-style="western"><surname>Song</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Exploring the role of knowledge graph-based RAG in Japanese medical question answering with small-scale LLMs</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 15, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2504.10982</pub-id></nlm-citation></ref><ref id="ref146"><label>146</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Han</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Ge</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>C</given-names> </name></person-group><article-title>Knowledge-guided large language model for automatic pediatric dental record understanding and safe antibiotic recommendation</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 9, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2512.09127</pub-id></nlm-citation></ref><ref id="ref147"><label>147</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Keerthana</surname><given-names>G</given-names> </name><name name-style="western"><surname>Gupta</surname><given-names>M</given-names> </name></person-group><article-title>CLI-RAG: a retrieval-augmented framework for clinically structured and context aware text generation with LLMs</article-title><source>arXiv</source><comment>Preprint posted online on  Jul 9, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2507.06715</pub-id></nlm-citation></ref><ref id="ref148"><label>148</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Lai</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ive</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Development and evaluation of HopeBot: an LLM-based chatbot for structured and interactive PHQ-9 depression screening</article-title><source>arXiv</source><comment>Preprint posted online on  Jul 8, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2507.05984</pub-id></nlm-citation></ref><ref id="ref149"><label>149</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Zhu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Guo</surname><given-names>J</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Gu</surname><given-names>P</given-names> </name><name name-style="western"><surname>Shu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>D</given-names> </name></person-group><article-title>Explainable interictal epileptiform discharge detection method based on scalp EEG and retrieval-augmented generation</article-title><source>arXiv</source><comment>Preprint posted online on  Feb 15, 2026</comment><pub-id pub-id-type="doi">10.48550/arXiv.2602.14170</pub-id></nlm-citation></ref><ref id="ref150"><label>150</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Mo</surname><given-names>G</given-names> </name><name name-style="western"><surname>Raman</surname><given-names>NJ</given-names> </name><name name-style="western"><surname>Chai</surname><given-names>M</given-names> </name><etal/></person-group><article-title>PeerCoPilot: a language model-powered assistant for behavioral health organizations</article-title><conf-name>AAAI&#x2019;26: AAAI Conference on Artificial Intelligence</conf-name><conf-date>Jan 20-27, 2026</conf-date><pub-id pub-id-type="doi">10.1609/aaai.v40i47.41476</pub-id></nlm-citation></ref><ref id="ref151"><label>151</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Shahnawaz</surname><given-names>A</given-names> </name><name name-style="western"><surname>Shafique</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Mustafa</surname><given-names>M</given-names> </name></person-group><article-title>Designing around stigma: human-centered LLMs for menstrual health</article-title><access-date>2026-07-06</access-date><conf-name>CHI &#x2019;26: Proceedings of the 2026 CHI Conference on Human Factors in Computing Systems</conf-name><conf-date>Apr 13-17, 2026</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/proceedings/10.1145/3772318">https://dl.acm.org/doi/proceedings/10.1145/3772318</ext-link></comment><pub-id pub-id-type="doi">10.1145/3772318.3791318</pub-id></nlm-citation></ref><ref id="ref152"><label>152</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Ahalpara</surname><given-names>TJ</given-names> </name></person-group><article-title>Tell Me: an LLM-powered mental well-being assistant with RAG, synthetic dialogue generation, and agentic planning</article-title><source>arXiv</source><comment>Preprint posted online on  Nov 18, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2511.14445</pub-id></nlm-citation></ref><ref id="ref153"><label>153</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Boumans</surname><given-names>R</given-names> </name><name name-style="western"><surname>Cramer</surname><given-names>L</given-names> </name><name name-style="western"><surname>van de Poll</surname><given-names>S</given-names> </name><name name-style="western"><surname>Vermeulen</surname><given-names>H</given-names> </name></person-group><article-title>A feasibility study on usability and trust among population groups of a medical avatar supported by large language models with retrieval augmented generation</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 17, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2510.15531</pub-id></nlm-citation></ref><ref id="ref154"><label>154</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Parmanto</surname><given-names>B</given-names> </name><name name-style="western"><surname>Aryoyudanta</surname><given-names>B</given-names> </name><name name-style="western"><surname>Soekinto</surname><given-names>TW</given-names> </name><etal/></person-group><article-title>A reliable and accessible caregiving language model (CaLM) to support tools for caregivers: development and evaluation study</article-title><source>JMIR Form Res</source><year>2024</year><month>07</month><day>31</day><volume>8</volume><fpage>e54633</fpage><pub-id pub-id-type="doi">10.2196/54633</pub-id><pub-id pub-id-type="medline">39083337</pub-id></nlm-citation></ref><ref id="ref155"><label>155</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hang</surname><given-names>CN</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>PD</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>CW</given-names> </name></person-group><article-title>TrumorGPT: graph-based retrieval-augmented large language model for fact-checking</article-title><source>IEEE Trans Artif Intell</source><year>2025</year><volume>6</volume><issue>11</issue><fpage>3148</fpage><lpage>3162</lpage><pub-id pub-id-type="doi">10.1109/TAI.2025.3567369</pub-id></nlm-citation></ref><ref id="ref156"><label>156</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Gu</surname><given-names>D</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>M</given-names> </name><name name-style="western"><surname>Metaxas</surname><given-names>D</given-names> </name></person-group><article-title>RadAlign: advancing radiology report generation with vision-language concept alignment</article-title><conf-name>Medical Image Computing and Computer Assisted Intervention &#x2013; MICCAI 2025</conf-name><conf-date>Sep 23-27, 2025</conf-date><pub-id pub-id-type="doi">10.1007/978-3-032-04981-0_46</pub-id></nlm-citation></ref><ref id="ref157"><label>157</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Yi</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Xiao</surname><given-names>T</given-names> </name><name name-style="western"><surname>Albert</surname><given-names>MV</given-names> </name></person-group><article-title>A multimodal multi-agent framework for radiology report generation</article-title><source>arXiv</source><comment>Preprint posted online on  May 14, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2505.09787</pub-id></nlm-citation></ref><ref id="ref158"><label>158</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Yoon</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>DJ</given-names> </name><name name-style="western"><surname>Park</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>YO</given-names> </name></person-group><article-title>Retrieval augmented large language model system for comprehensive drug contraindications</article-title><source>Health Inf Sci Syst</source><year>2026</year><month>01</month><day>11</day><volume>14</volume><issue>1</issue><fpage>26</fpage><pub-id pub-id-type="doi">10.1007/s13755-025-00420-z</pub-id><pub-id pub-id-type="medline">41531551</pub-id></nlm-citation></ref><ref id="ref159"><label>159</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Ting</surname><given-names>LPY</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zeng</surname><given-names>YH</given-names> </name><name name-style="western"><surname>Lim</surname><given-names>YJ</given-names> </name><name name-style="western"><surname>Chuang</surname><given-names>KT</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>H</given-names> </name></person-group><article-title>Leaps beyond the seen: reinforced reasoning augmented generation for clinical notes</article-title><source>arXiv</source><comment>Preprint posted online on  Jun 3, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2506.05386</pub-id></nlm-citation></ref><ref id="ref160"><label>160</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Johno</surname><given-names>H</given-names> </name><name name-style="western"><surname>Johno</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Amakawa</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Enhancing pancreatic cancer staging with large language models: the role of retrieval-augmented generation</article-title><source>Radiol Phys Technol</source><year>2026</year><month>06</month><volume>19</volume><issue>2</issue><fpage>593</fpage><lpage>603</lpage><pub-id pub-id-type="doi">10.1007/s12194-026-01026-0</pub-id><pub-id pub-id-type="medline">41784908</pub-id></nlm-citation></ref><ref id="ref161"><label>161</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>He</surname><given-names>J</given-names> </name><name name-style="western"><surname>Guo</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Lam</surname><given-names>LK</given-names> </name><etal/></person-group><article-title>OpenTCM: a GraphRAG-empowered LLM-based system for traditional chinese medicine knowledge retrieval and diagnosis</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 28, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2504.20118</pub-id></nlm-citation></ref><ref id="ref162"><label>162</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhao</surname><given-names>YF</given-names> </name><name name-style="western"><surname>Bove</surname><given-names>A</given-names> </name><name name-style="western"><surname>Thompson</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Generative AI Is not ready for clinical use in patient education for lower back pain patients, even with retrieval-augmented generation</article-title><source>AMIA Jt Summits Transl Sci Proc</source><year>2025</year><volume>2025</volume><fpage>644</fpage><lpage>653</lpage><pub-id pub-id-type="medline">40502233</pub-id></nlm-citation></ref><ref id="ref163"><label>163</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Madrid-Garc&#x00ED;a</surname><given-names>A</given-names> </name><name name-style="western"><surname>Benavent</surname><given-names>D</given-names> </name><name name-style="western"><surname>Plasencia-Rodr&#x00ED;guez</surname><given-names>C</given-names> </name><name name-style="western"><surname>Rosales-Rosado</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Merino-Barbancho</surname><given-names>B</given-names> </name><name name-style="western"><surname>Freites-N&#x00FA;&#x00F1;ez</surname><given-names>D</given-names> </name></person-group><article-title>Optimising the clinical application of rheumatology guidelines using large language models: a retrieval-augmented generation framework integrating EULAR and ACR recommendations</article-title><source>EULAR Rheumatol Open</source><year>2025</year><month>10</month><volume>1</volume><issue>3</issue><fpage>228</fpage><lpage>236</lpage><pub-id pub-id-type="doi">10.1016/j.ero.2025.08.001</pub-id><pub-id pub-id-type="medline">42368617</pub-id></nlm-citation></ref><ref id="ref164"><label>164</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Saidu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Wall</surname><given-names>J</given-names> </name></person-group><article-title>Retrieval-augmented large language model for clinical decision support with a medical knowledge graph</article-title><source>Electronics (Basel)</source><year>2026</year><volume>15</volume><issue>3</issue><fpage>555</fpage><pub-id pub-id-type="doi">10.3390/electronics15030555</pub-id></nlm-citation></ref><ref id="ref165"><label>165</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Felde</surname><given-names>S</given-names> </name><name name-style="western"><surname>Buchkremer</surname><given-names>R</given-names> </name><name name-style="western"><surname>Chehab</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Low-energy small language models with retrieval-augmented generation can surpass large-model performance in rheumatology</article-title><source>Front Med (Lausanne)</source><year>2026</year><month>05</month><day>8</day><volume>13</volume><fpage>1817215</fpage><pub-id pub-id-type="doi">10.3389/fmed.2026.1817215</pub-id><pub-id pub-id-type="medline">42180760</pub-id></nlm-citation></ref><ref id="ref166"><label>166</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jeon</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Youn</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Kang</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Hierarchical RAG enhances a pharmacogenomic AI assistant in guideline related queries</article-title><source>Comput Biol Med</source><year>2026</year><month>01</month><day>1</day><volume>200</volume><fpage>111323</fpage><pub-id pub-id-type="doi">10.1016/j.compbiomed.2025.111323</pub-id><pub-id pub-id-type="medline">41319469</pub-id></nlm-citation></ref><ref id="ref167"><label>167</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>K</given-names> </name><name name-style="western"><surname>Cheng</surname><given-names>D</given-names> </name><name name-style="western"><surname>Yuan</surname><given-names>L</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>W</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>K</given-names> </name></person-group><article-title>Evaluation of large language models and retrieval-augmented generation for clinical reasoning in pediatric myopia: a 50-case real-world study</article-title><source>Sci Rep</source><year>2026</year><month>05</month><day>7</day><volume>16</volume><issue>1</issue><fpage>21059</fpage><pub-id pub-id-type="doi">10.1038/s41598-026-51205-7</pub-id><pub-id pub-id-type="medline">42098248</pub-id></nlm-citation></ref><ref id="ref168"><label>168</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Komenda</surname><given-names>A</given-names> </name><name name-style="western"><surname>Makowski</surname><given-names>M</given-names> </name><name name-style="western"><surname>Can</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Development and evaluation of a retrieval-augmented generation system for radiology guidelines</article-title><source>J Imaging Inform Med</source><year>2026</year><month>02</month><day>12</day><pub-id pub-id-type="doi">10.1007/s10278-025-01835-6</pub-id><pub-id pub-id-type="medline">41680573</pub-id></nlm-citation></ref><ref id="ref169"><label>169</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Luan</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Cheng</surname><given-names>S</given-names> </name><etal/></person-group><article-title>A multi-layer retrieval-augmented large language model framework for enhancing hypertension education</article-title><source>Hypertens Res</source><year>2026</year><month>04</month><volume>49</volume><issue>4</issue><fpage>1428</fpage><lpage>1440</lpage><pub-id pub-id-type="doi">10.1038/s41440-025-02481-9</pub-id><pub-id pub-id-type="medline">41501362</pub-id></nlm-citation></ref><ref id="ref170"><label>170</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Phan</surname><given-names>E</given-names> </name><name name-style="western"><surname>Velmovitsky</surname><given-names>P</given-names> </name><name name-style="western"><surname>Pham</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Sanner</surname><given-names>S</given-names> </name></person-group><article-title>Retrieval-augmented generation for medical question answering on a heart failure dataset: performance analysis</article-title><source>JMIR Form Res</source><year>2026</year><month>02</month><day>26</day><volume>10</volume><fpage>e84932</fpage><pub-id pub-id-type="doi">10.2196/84932</pub-id><pub-id pub-id-type="medline">41747226</pub-id></nlm-citation></ref><ref id="ref171"><label>171</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Baseri Saadi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ver Berne</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cavalcante Fontenele</surname><given-names>R</given-names> </name><name name-style="western"><surname>Claes</surname><given-names>P</given-names> </name><name name-style="western"><surname>Jacobs</surname><given-names>R</given-names> </name></person-group><article-title>JADE: jawbone lesion diagnosis and decision supporting system</article-title><source>Dentomaxillofac Radiol</source><year>2026</year><month>07</month><day>1</day><volume>55</volume><issue>5</issue><fpage>497</fpage><lpage>507</lpage><pub-id pub-id-type="doi">10.1093/dmfr/twag017</pub-id><pub-id pub-id-type="medline">41872013</pub-id></nlm-citation></ref><ref id="ref172"><label>172</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Song</surname><given-names>JK</given-names> </name><name name-style="western"><surname>Youk</surname><given-names>DB</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>H</given-names> </name><name name-style="western"><surname>Hwang</surname><given-names>SH</given-names> </name></person-group><article-title>Multimodal knowledge graph-guided RAG-LLM for clinical decision support in pediatric leukemia</article-title><source>Cancer Res Treat</source><year>2026</year><month>04</month><day>21</day><pub-id pub-id-type="doi">10.4143/crt.2026.0047</pub-id><pub-id pub-id-type="medline">42025216</pub-id></nlm-citation></ref><ref id="ref173"><label>173</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Kabak</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Erturkmen</surname><given-names>GBL</given-names> </name><name name-style="western"><surname>Gencturk</surname><given-names>M</given-names> </name></person-group><article-title>FHIR-RAG-MEDS: integrating HL7 FHIR with retrieval-augmented large language models for enhanced medical decision support</article-title><source>arXiv</source><comment>Preprint posted online on  Sep 9, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2509.07706</pub-id></nlm-citation></ref><ref id="ref174"><label>174</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Ma</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yue</surname><given-names>X</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wei</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Jia</surname><given-names>Z</given-names> </name></person-group><article-title>MGK-RAG: multi-granularity knowledge guided retrieval-augmented generation for radiology report</article-title><conf-name>WWW &#x2019;26</conf-name><conf-date>Apr 13-17, 2026</conf-date><pub-id pub-id-type="doi">10.1145/3774904.3792924</pub-id></nlm-citation></ref><ref id="ref175"><label>175</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>KM</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>M</given-names> </name></person-group><article-title>CPR-RAG: clinical prior-regularized retrieval for anatomy-aware 3D CT report generation</article-title><conf-name>Proceedings of the 64th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)</conf-name><conf-date>Jul 2-7, 2026</conf-date><pub-id pub-id-type="doi">10.18653/v1/2026.acl-long.1411</pub-id></nlm-citation></ref><ref id="ref176"><label>176</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Garcia-Font</surname><given-names>M</given-names> </name><name name-style="western"><surname>Dufey-Portilla</surname><given-names>N</given-names> </name><name name-style="western"><surname>Dur&#x00E1;n-Sindreu</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Evaluating retrieval-augmented large language models on external cervical resorption: a comparative study of Gemini and NotebookLM</article-title><source>J Endod</source><year>2026</year><month>02</month><volume>52</volume><issue>2</issue><fpage>300</fpage><lpage>306</lpage><pub-id pub-id-type="doi">10.1016/j.joen.2025.10.016</pub-id><pub-id pub-id-type="medline">41207474</pub-id></nlm-citation></ref><ref id="ref177"><label>177</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wong</surname><given-names>HS</given-names> </name><name name-style="western"><surname>Wong</surname><given-names>TK</given-names> </name></person-group><article-title>Multi-evidence clinical reasoning with retrieval-augmented generation for emergency triage: retrospective evaluation study</article-title><source>JMIR Med Inform</source><year>2026</year><month>01</month><day>26</day><volume>14</volume><fpage>e82026</fpage><pub-id pub-id-type="doi">10.2196/82026</pub-id><pub-id pub-id-type="medline">41587455</pub-id></nlm-citation></ref><ref id="ref178"><label>178</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thio</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lewis</surname><given-names>M</given-names> </name><name name-style="western"><surname>Denaxas</surname><given-names>S</given-names> </name><name name-style="western"><surname>Dobson</surname><given-names>RJB</given-names> </name></person-group><article-title>Unlocking electronic health records: a hybrid graph RAG approach to safe clinical AI for patient QA</article-title><source>Front Digit Health</source><year>2026</year><volume>8</volume><fpage>1780700</fpage><pub-id pub-id-type="doi">10.3389/fdgth.2026.1780700</pub-id><pub-id pub-id-type="medline">41890309</pub-id></nlm-citation></ref><ref id="ref179"><label>179</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zaki</surname><given-names>HA</given-names> </name><name name-style="western"><surname>Brea</surname><given-names>A</given-names> </name><name name-style="western"><surname>Parvataneni</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Retrieval-augmented language models for patient-centered periprocedural anticoagulation in interventional radiology</article-title><source>Cardiovasc Intervent Radiol</source><year>2026</year><month>06</month><volume>49</volume><issue>6</issue><fpage>1177</fpage><lpage>1187</lpage><pub-id pub-id-type="doi">10.1007/s00270-026-04428-0</pub-id><pub-id pub-id-type="medline">41917168</pub-id></nlm-citation></ref><ref id="ref180"><label>180</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Li</surname><given-names>D</given-names> </name><etal/></person-group><article-title>LLM-driven collaborative framework for knowledge-enhanced cancer pain assessment and management</article-title><source>NPJ Digit Med</source><year>2026</year><month>01</month><day>19</day><volume>9</volume><issue>1</issue><fpage>180</fpage><pub-id pub-id-type="doi">10.1038/s41746-026-02362-6</pub-id><pub-id pub-id-type="medline">41554973</pub-id></nlm-citation></ref><ref id="ref181"><label>181</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>D</given-names> </name><name name-style="western"><surname>Yoo</surname><given-names>S</given-names> </name><name name-style="western"><surname>Jeong</surname><given-names>O</given-names> </name></person-group><article-title>MedSumGraph: enhancing GraphRAG for medical QA with summarization and optimized prompts</article-title><source>Artif Intell Med</source><year>2026</year><month>02</month><volume>172</volume><fpage>103311</fpage><pub-id pub-id-type="doi">10.1016/j.artmed.2025.103311</pub-id><pub-id pub-id-type="medline">41319399</pub-id></nlm-citation></ref><ref id="ref182"><label>182</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lopez</surname><given-names>I</given-names> </name><name name-style="western"><surname>Swaminathan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Vedula</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Clinical entity augmented retrieval for clinical information extraction</article-title><source>NPJ Digit Med</source><year>2025</year><month>01</month><day>19</day><volume>8</volume><issue>1</issue><fpage>45</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01377-1</pub-id><pub-id pub-id-type="medline">39828800</pub-id></nlm-citation></ref><ref id="ref183"><label>183</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nanua</surname><given-names>S</given-names> </name><name name-style="western"><surname>Steward</surname><given-names>R</given-names> </name><name name-style="western"><surname>Neely</surname><given-names>B</given-names> </name><name name-style="western"><surname>Datto</surname><given-names>M</given-names> </name><name name-style="western"><surname>Youens</surname><given-names>K</given-names> </name></person-group><article-title>Retrieval-augmented generation for interpreting clinical laboratory regulations using large language models</article-title><source>J Pathol Inform</source><year>2025</year><month>11</month><volume>19</volume><fpage>100520</fpage><pub-id pub-id-type="doi">10.1016/j.jpi.2025.100520</pub-id><pub-id pub-id-type="medline">41244595</pub-id></nlm-citation></ref><ref id="ref184"><label>184</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xie</surname><given-names>W</given-names> </name><name name-style="western"><surname>Song</surname><given-names>X</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Retrieval&#x2011;augmented large language models for depression screening and suicide risk stratification</article-title><source>BMC Psychiatry</source><year>2026</year><month>03</month><day>31</day><volume>26</volume><issue>1</issue><fpage>386</fpage><pub-id pub-id-type="doi">10.1186/s12888-026-07988-0</pub-id><pub-id pub-id-type="medline">41917861</pub-id></nlm-citation></ref><ref id="ref185"><label>185</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>W</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>W</given-names> </name><etal/></person-group><article-title>A survey on hallucination in large language models: principles, taxonomy, challenges, and open questions</article-title><source>ACM Trans Inf Syst</source><year>2025</year><month>03</month><day>31</day><volume>43</volume><issue>2</issue><fpage>1</fpage><lpage>55</lpage><pub-id pub-id-type="doi">10.1145/3703155</pub-id></nlm-citation></ref><ref id="ref186"><label>186</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wallat</surname><given-names>J</given-names> </name><name name-style="western"><surname>Heuss</surname><given-names>M</given-names> </name><name name-style="western"><surname>de Rijke</surname><given-names>M</given-names> </name><name name-style="western"><surname>Anand</surname><given-names>A</given-names> </name></person-group><article-title>Correctness is not faithfulness in retrieval augmented generation attributions</article-title><conf-name>ICTIR '25: Proceedings of the 2025 International ACM SIGIR Conference on Innovative Concepts and Theories in Information Retrieval (ICTIR)</conf-name><conf-date>Jul 18, 2025</conf-date><conf-loc>Padua, Italy</conf-loc><fpage>22</fpage><lpage>32</lpage><pub-id pub-id-type="doi">10.1145/3731120.3744592</pub-id></nlm-citation></ref><ref id="ref187"><label>187</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Min</surname><given-names>S</given-names> </name><name name-style="western"><surname>Krishna</surname><given-names>K</given-names> </name><name name-style="western"><surname>Lyu</surname><given-names>X</given-names> </name><etal/></person-group><article-title>FActScore: fine-grained atomic evaluation of factual precision in long form text generation</article-title><conf-name>Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Dec 6-10, 2023</conf-date><conf-loc>Singapore</conf-loc><fpage>12076</fpage><lpage>12100</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.emnlp-main.741</pub-id></nlm-citation></ref><ref id="ref188"><label>188</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Clark</surname><given-names>E</given-names> </name><name name-style="western"><surname>August</surname><given-names>T</given-names> </name><name name-style="western"><surname>Serrano</surname><given-names>S</given-names> </name><name name-style="western"><surname>Haduong</surname><given-names>N</given-names> </name><name name-style="western"><surname>Gururangan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>NA</given-names> </name></person-group><article-title>All that&#x2019;s &#x2018;human&#x2019; is not gold: evaluating human evaluation of generated text</article-title><conf-name>Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers)</conf-name><conf-date>Aug 1-6, 2021</conf-date><conf-loc>Online</conf-loc><fpage>7282</fpage><lpage>7296</lpage><pub-id pub-id-type="doi">10.18653/v1/2021.acl-long.565</pub-id></nlm-citation></ref><ref id="ref189"><label>189</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zheng</surname><given-names>L</given-names> </name><name name-style="western"><surname>Chiang</surname><given-names>WL</given-names> </name><name name-style="western"><surname>Sheng</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena</article-title><access-date>2026-07-29</access-date><conf-name>37th Conference on Neural Information Processing Systems (NeurIPS 2023) Track on Datasets and Benchmarks</conf-name><conf-date>Dec 10-16, 2023</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.proceedings.com/content/075/075280-2020open.pdf">https://www.proceedings.com/content/075/075280-2020open.pdf</ext-link></comment><pub-id pub-id-type="doi">10.52202/075280-2020</pub-id></nlm-citation></ref><ref id="ref190"><label>190</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Razavi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Soltangheis</surname><given-names>M</given-names> </name><name name-style="western"><surname>Arabzadeh</surname><given-names>N</given-names> </name><name name-style="western"><surname>Salamat</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zihayat</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bagheri</surname><given-names>E</given-names> </name></person-group><article-title>Benchmarking prompt sensitivity in large language models</article-title><conf-name>Advances in Information Retrieval: 47th European Conference on Information Retrieval, ECIR 2025</conf-name><conf-date>Apr 6-10, 2025</conf-date><pub-id pub-id-type="doi">10.1007/978-3-031-88714-7_29</pub-id></nlm-citation></ref><ref id="ref191"><label>191</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Zhao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>D</given-names> </name><name name-style="western"><surname>Kung</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mi</surname><given-names>H</given-names> </name><etal/></person-group><article-title>One token to fool LLM-as-a-judge</article-title><source>arXiv</source><comment>Preprint posted online on  Jul 11, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2507.08794</pub-id></nlm-citation></ref><ref id="ref192"><label>192</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><name name-style="western"><surname>Dong</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>J</given-names> </name><etal/></person-group><article-title>LLMs-as-judges: a comprehensive survey on LLM-based evaluation methods</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 7, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2412.05579</pub-id></nlm-citation></ref><ref id="ref193"><label>193</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Prasad</surname><given-names>A</given-names> </name><name name-style="western"><surname>Stengel-Eskin</surname><given-names>E</given-names> </name><name name-style="western"><surname>Bansal</surname><given-names>M</given-names> </name></person-group><article-title>Retrieval-augmented generation with conflicting evidence</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 17, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2504.13079</pub-id></nlm-citation></ref><ref id="ref194"><label>194</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Han</surname><given-names>T</given-names> </name><name name-style="western"><surname>Kumar</surname><given-names>A</given-names> </name><name name-style="western"><surname>Agarwal</surname><given-names>C</given-names> </name><name name-style="western"><surname>Lakkaraju</surname><given-names>H</given-names> </name></person-group><article-title>MedSafetyBench: evaluating and improving the medical safety of large language models</article-title><conf-name>NIPS &#x2019;24: Proceedings of the 38th International Conference on Neural Information Processing Systems</conf-name><conf-date>Dec 10-15, 2024</conf-date><pub-id pub-id-type="doi">10.52202/079017-1054</pub-id></nlm-citation></ref><ref id="ref195"><label>195</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Savage</surname><given-names>T</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Gallo</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Large language model uncertainty measurement and calibration for medical diagnosis and treatment</article-title><source>medRxiv</source><comment>Preprint posted online on  Jun 10, 2024</comment><pub-id pub-id-type="doi">10.1101/2024.06.06.24308399</pub-id></nlm-citation></ref><ref id="ref196"><label>196</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sittig</surname><given-names>DF</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>H</given-names> </name></person-group><article-title>A new sociotechnical model for studying health information technology in complex adaptive healthcare systems</article-title><source>Qual Saf Health Care</source><year>2010</year><month>10</month><volume>19 Suppl 3</volume><issue>Suppl 3</issue><fpage>i68</fpage><lpage>74</lpage><pub-id pub-id-type="doi">10.1136/qshc.2010.042085</pub-id><pub-id pub-id-type="medline">20959322</pub-id></nlm-citation></ref><ref id="ref197"><label>197</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Barnett</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kurniawan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Thudumu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Brannelly</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Abdelrazek</surname><given-names>M</given-names> </name></person-group><article-title>Seven failure points when engineering a retrieval augmented generation system</article-title><conf-name>CAIN 2024</conf-name><conf-date>Apr 14-15, 2024</conf-date><pub-id pub-id-type="doi">10.1145/3644815.3644945</pub-id></nlm-citation></ref><ref id="ref198"><label>198</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>van der Vorst</surname><given-names>JP</given-names> </name><name name-style="western"><surname>Smit</surname><given-names>JM</given-names> </name><name name-style="western"><surname>van de Sande</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Importance of model governance in clinical AI models: case study on the relevance of data drift detection</article-title><source>BMJ Digit Health</source><year>2025</year><month>07</month><volume>1</volume><issue>1</issue><fpage>e000046</fpage><pub-id pub-id-type="doi">10.1136/bmjdhai-2025-000046</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Complete database-specific search strategies.</p><media xlink:href="jmir_v28i1e90046_app1.docx" xlink:title="DOCX File, 34 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Study-level core coding tables and source data for the evidence-and-gap map.</p><media xlink:href="jmir_v28i1e90046_app2.docx" xlink:title="DOCX File, 142 KB"/></supplementary-material><supplementary-material id="app3"><label>Checklist 1</label><p>PRISMA-ScR checklist.</p><media xlink:href="jmir_v28i1e90046_app3.docx" xlink:title="DOCX File, 44 KB"/></supplementary-material><supplementary-material id="app4"><label>Checklist 2</label><p>PRISMA-S checklist.</p><media xlink:href="jmir_v28i1e90046_app4.docx" xlink:title="DOCX File, 29 KB"/></supplementary-material></app-group></back></article>