<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e92090</article-id><article-id pub-id-type="doi">10.2196/92090</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Improving Reliability and Explainability of Medical Question Answering Through Atomic Fact-Checking in Retrieval-Augmented Large Language Models: Creation and Validation Study</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Vladika</surname><given-names>Juraj</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Domres</surname><given-names>Annika</given-names></name><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Nguyen</surname><given-names>Mai</given-names></name><degrees>Dr med</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Moser</surname><given-names>Rebecca</given-names></name><degrees>Dr med</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Nano</surname><given-names>Jana</given-names></name><degrees>Dsc, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Busch</surname><given-names>Felix</given-names></name><degrees>Dr med</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Adams</surname><given-names>Lisa</given-names></name><degrees>Prof Dr med</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Bressem</surname><given-names>Keno K</given-names></name><degrees>Prof Dr med</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Bernhardt</surname><given-names>Denise</given-names></name><degrees>Dr med</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Combs</surname><given-names>Stephanie E</given-names></name><degrees>Prof Dr med</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff7">7</xref><xref ref-type="aff" rid="aff8">8</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Borm</surname><given-names>Kai</given-names></name><degrees>Prof Dr med</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Matthes</surname><given-names>Florian</given-names></name><degrees>Prof Dr</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Peeken</surname><given-names>Jan C</given-names></name><degrees>Prof Dr med, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff7">7</xref><xref ref-type="aff" rid="aff8">8</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Computer Science, TUM School of Computation Information and Technology, Technical University of Munich</institution><addr-line>Garching</addr-line><addr-line>Bavaria</addr-line><country>Germany</country></aff><aff id="aff2"><institution>Department of Radiation Oncology, TUM University Hospital Rechts der Isar, TUM School of Medicine and Health, Technical University of Munich</institution><addr-line>Ismaninger Str. 22</addr-line><addr-line>Munich</addr-line><addr-line>Bavaria</addr-line><country>Germany</country></aff><aff id="aff3"><institution>Department of Diagnostic and Interventional Radiology, TUM University Hospital Rechts der Isar, TUM School of Medicine and Health, Technical University of Munich</institution><addr-line>Munich</addr-line><country>Germany</country></aff><aff id="aff4"><institution>National Center for Tumor Diseases West</institution><addr-line>Essen</addr-line><country>Germany</country></aff><aff id="aff5"><institution>Institute of Artificial Intelligence in Medicine, University Hospital Essen</institution><addr-line>Essen</addr-line><country>Germany</country></aff><aff id="aff6"><institution>Department of Diagnostic and Interventional Radiology and Neuroradiology, University Hospital Essen</institution><addr-line>Essen</addr-line><country>Germany</country></aff><aff id="aff7"><institution>Institute of Radiation Medicine (IRM), Helmholtz Zentrum M&#x00FC;nchen (HMGU)</institution><addr-line>Munich</addr-line><addr-line>Bavaria</addr-line><country>Germany</country></aff><aff id="aff8"><institution>German Consortium for Translational Cancer Research (DKTK), Partner Site Munich</institution><addr-line>Munich</addr-line><addr-line>Bavaria</addr-line><country>Germany</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Bravo</surname><given-names>Fernanda</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Cho</surname><given-names>Jeonghun</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Masanneck</surname><given-names>Lars</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Jan C Peeken, Prof Dr med, PhD, Department of Radiation Oncology, TUM University Hospital Rechts der Isar, TUM School of Medicine and Health, Technical University of Munich, Ismaningerstr 22, Munich, Bavaria, Germany, 49 089 4140-4501; <email>jan.peeken@tum.de</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>21</day><month>9</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e92090</elocation-id><history><date date-type="received"><day>27</day><month>01</month><year>2026</year></date><date date-type="rev-recd"><day>29</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>29</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Juraj Vladika, Annika Domres, Mai Nguyen, Rebecca Moser, Jana Nano, Felix Busch, Lisa Adams, Keno K Bressem, Denise Bernhardt, Stephanie E Combs, Kai Borm, Florian Matthes, Jan C Peeken. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 21.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e92090"/><abstract><sec><title>Background</title><p>Large language models (LLMs) exhibit extensive medical knowledge but are prone to hallucinations and show low fact-level explainability, limiting clinical adoption and regulatory compliance. Existing approaches, such as retrieval-augmented generation, partially address these issues by grounding answers in source documents; however, the aforementioned problems persist.</p></sec><sec><title>Objective</title><p>We propose the application of an atomic fact-checking framework designed to enhance the reliability and explainability of LLMs in medical long-form question answering. By decomposing generated answers into discrete atomic facts and verifying each against an authoritative knowledge base of medical guidelines, this approach enables precise identification and correction of incorrect statements, alongside explicit linkage to supporting literature.</p></sec><sec sec-type="methods"><title>Methods</title><p>The fact-checking algorithm operates within a retrieval-augmented generation framework: LLM-generated answers are decomposed into atomic facts (smallest and self-contained information units), each of which is assessed and corrected if FALSE. To determine an optimal strategy, the validation&#x2013;question and answer (Q&#x0026;A) set on prostate cancer treatment was tested under varying instructions. An extensive evaluation, including multireader assessments by human medical experts and the automated open Q&#x0026;A benchmark AMEGA (Autonomous Medical Evaluation for Guideline Adherence), was conducted for the final pipeline. In addition to another radiooncologic test&#x2013;Q&#x0026;A set, anonymized real-world tumor board cases and an independent, established neurology-Q&#x0026;A set were used. Given their transparency and accessibility advantages, we compared various open-source models in pairs of generalist models and their medical fine-tuned counterparts, with regard to performance and improvements by fact-checking.</p></sec><sec sec-type="results"><title>Results</title><p>The framework significantly reduced hallucinations and inaccuracies. Medical expert assessment and automated benchmarks demonstrated significant improvements in factual accuracy, achieving up to a 50% overall answer improvement and an 80% hallucination detection rate. Notably, the observed gain was strongest in real tumor-board questions&#x2014;the most challenging dataset. Additionally, the framework achieved high explainability by tracing each atomic fact back to the most relevant chunks from the database, providing a granular, transparent explanation of the generated responses.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>To conclude, we present the application of an atomic fact-checking algorithm to medical Q&#x0026;A. It identifies factual inaccuracies and hallucinations in LLM-generated answers, achieving the greatest gains on clinically realistic, complex questions. Correction via fact-checking improves the overall answer quality while achieving fact-wise explainability, paving the way for more credible clinical use of LLMs.</p></sec></abstract><kwd-group><kwd>large language model</kwd><kwd>LLM</kwd><kwd>retrieval-augmented generation</kwd><kwd>RAG</kwd><kwd>fact-checking</kwd><kwd>atomic fact-checking</kwd><kwd>atomic fact</kwd><kwd>open-source</kwd><kwd>autoevaluation</kwd><kwd>hallucination</kwd><kwd>backtracing</kwd><kwd>prompt engineering</kwd><kwd>radiation oncology</kwd><kwd>medical fine-tuned</kwd><kwd>rubrics</kwd><kwd>question and answer</kwd><kwd>medical Q&#x0026;A</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Large language models (LLMs) exhibit extensive medical knowledge [<xref ref-type="bibr" rid="ref1">1</xref>]. However, LLMs are prone to hallucinations that may lead to harmful medical advice, and their often inaccurate citations reduce overall explainability [<xref ref-type="bibr" rid="ref2">2</xref>]. This limits clinical use and complicates medical product certifications [<xref ref-type="bibr" rid="ref3">3</xref>]. Current methods, such as retrieval-augmented generation (RAG), partially address these issues by grounding answers in source documents [<xref ref-type="bibr" rid="ref4">4</xref>].</p><p>In RAG, the most common setup is to divide source texts into discrete chunks, embed them into a vector space, and retrieve as needed to ground LLM responses in updatable, authoritative information. Prior research shows that RAG can improve answer quality in medical question and answer (Q&#x0026;A) [<xref ref-type="bibr" rid="ref5">5</xref>]. Nevertheless, answers can still contain hallucinations, that is statements that are factually incorrect and contradict established knowledge. Additionally, fact-by-fact explainability of answers remains low, especially for complex medical queries. Existing approaches do not address the need for validating, backtracing, and correcting each individual claim within a long-form response [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>].</p><p>Methods to increase the factuality of LLM outputs can involve additional pretraining or fine-tuning of models, which are computationally expensive and impossible for closed-source models. Hence, post hoc methods are emerging, where LLMs self-correct their responses only after they are generated [<xref ref-type="bibr" rid="ref8">8</xref>]. A promising method is automated fact-checking, which detects individual facts from the generated response that contain information contradicting the authoritative knowledge, then rewrites these facts and the final response using correct information. Originating from journalism, where it is performed manually, fact-checking is increasingly used for hallucination correction [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. However, existing approaches mostly focus on the encyclopedic and news domains and are underexplored for medical applications [<xref ref-type="bibr" rid="ref11">11</xref>].</p><p>We define an atomic fact as the smallest self-contained and verifiable unit of information in a response generated by an LLM [<xref ref-type="bibr" rid="ref10">10</xref>] (eg, &#x201C;Trastuzumab is indicated for HER2-positive breast cancer&#x201D;). We developed a framework that decomposes responses into atomic facts, each of which is independently verified against an authoritative vector database. This approach enables targeted correction of errors and direct tracing to the source literature, thereby improving the factual accuracy and explainability of medical Q&#x0026;A. To the best of our knowledge, this is the first application of such an &#x201C;atomic fact-checking&#x201D; approach to the medical domain.</p><p>Beyond the provision of correct, up-to-date information, which represents a fundamental prerequisite for clinical adaptation, successful adoption of LLMs also depends on the users&#x2019; trust in both the generated answer and the underlying technical infrastructure.</p><p>Increasing scientific attention is being directed toward open-source models, resulting in capability improvements approaching those of proprietary models. This trend is driven by several advantages of open-source approaches, including greater transparency and enhanced data privacy, which favor their use [<xref ref-type="bibr" rid="ref12">12</xref>]. Accordingly, we evaluated a variety of open-source models with respect to their performance and the improvements through fact-checking.</p><p>We hypothesize that (1) integrating an atomic fact-checking framework into an RAG pipeline improves the factual quality of medical Q&#x0026;A responses while enabling fact-level traceability; (2) the magnitude of improvement achieved via fact-checking depends on the underlying model; and (3) atomic fact-checking yields performance improvements across different medical use cases, including real-world tumor board cases.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Atomic Fact-Checking Framework</title><p>The LLM-TRIPOD (Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis) checklist is provided as <xref ref-type="supplementary-material" rid="app2">Checklist 1</xref>. Our fact-checking framework consists of five steps, as shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>: (1) generate an initial RAG-based response to the question; (2) split the response into atomic facts; (3) determine the veracity of each fact (categories: &#x201C;TRUE&#x201D; and &#x201C;FALSE&#x201D;) based on newly retrieved chunks; (4) rewrite the facts detected as incorrect and loop through steps 3 and 4 until all facts are correct, or for a maximum of 3 iterations; (5) rewrite the full response by incorporating the rewritten facts.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Architecture diagram illustrating the atomic fact-checking process. RAG: retrieval-augmented generation.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92090_fig01.png"/></fig><p>All 5 steps are implemented using dedicated LLM prompts, provided in Table S1 of <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. For steps 1 to 4, in-context learning with 4 expert-annotated examples per prompt is used to enforce consistent atomic fact extraction and verdicting. All prompts and examples were created by a medical expert (JCP, 9 y of experience in radiation oncology). Prompts were refined until the performance was deemed satisfactory on a validation set; this included increasing the number of few-shot examples from 0 to 1 and finally to 4. The number of retrieved chunks was set to 7 because in the initial testing it provided the best balance between sufficient useful information and minimizing the noise from irrelevant chunks. An analysis using different numbers of top <italic>k</italic> chunks (3, 5, 7, and 10) is shown in Table S12 of <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>The knowledge base consisted of curated oncological guideline documents for prostate and breast cancer (see Table S5 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). All PDF documents were converted into plain text using the open-source library PaperMage, which works well with visually rich scientific documents. The text was chunked into overlapping segments of 512 tokens (100-token overlap) and embedded using S-PubMedBERT [<xref ref-type="bibr" rid="ref13">13</xref>], a transformer model pretrained on PubMed abstracts, giving it an increased semantic understanding of medical concepts, making it highly suitable for our medical use case. Chunks were stored in a ChromaDB vector database. For evidence retrieval, cosine similarity was used to select the 7 most relevant chunks for the question (in step 1) and then the 7 most similar ones for each atomic fact (step 3). All LLM generations were performed using GPT-4o (gpt-4o-2024-11-20; OpenAI) via the OpenAI API with a temperature set to 0 to ensure more deterministic outputs and reproducibility.</p><p>While the knowledge base serves as the primary source for both answer creation and verification, the input question itself is also incorporated as a reference source during the fact-checking process. This allows for the verification of patient-specific information included in the query that is not represented in guideline documents. Consequently, patient-specific facts are not incorrectly classified as FALSE only because they cannot be validated against the external knowledge base.</p><p>Two additional components of the pipeline were tested. Looping, where the results of one full fact-checking run (with atomic fact corrections) were used as the input to another full run, was tested to determine if it helps correct the facts that were still incorrect. Ensembling, where the fact-veracity prediction (&#x201C;TRUE&#x201D; or &#x201C;FALSE&#x201D;) was based not on just one prediction output but on multiple predictions using the same LLM and prompt, was used to determine whether fact-veracity prediction could be improved.</p></sec><sec id="s2-2"><title>Q&#x0026;A Benchmark</title><p>The main task of this study was long-form medical question-answering. Two Q&#x0026;A datasets were constructed by a medical expert (JCP), with each dataset comprising distinct questions about the diagnosis and treatment of prostate or breast cancer. Questions were based on the guideline content provided by the RAG system. They were categorized as fact-based (direct guideline knowledge) or patient-based (clinical vignettes), with varying complexity. To select the optimal prompting strategy, the first set, a validation set of 50 Q&#x0026;A pairs on prostate cancer, was used. The final configuration was tested on the second set, a separate test set of 60 Q&#x0026;A pairs (30 prostate and 30 breast cancer).</p><p>Asking questions on specialties other than prostate cancer, as in the validation set (implemented breast cancer Q&#x0026;A in the test set and an independent, established neurology-Q&#x0026;A set [<xref ref-type="bibr" rid="ref2">2</xref>] of 65 Q&#x0026;A items), ensured that no overfitting occurred.</p><p>To assess the framework in a realistic clinical setting, 40 anonymized real patient cases from a multidisciplinary tumor board were included. A total of 215 Q&#x0026;As were used, all of which were checked by human expert evaluators.</p><p>All Q&#x0026;A sets are available in the GitHub (Microsoft) repository [<xref ref-type="bibr" rid="ref14">14</xref>]. An overview of all Q&#x0026;As used and their respective questions is available in Table S10 of the <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-3"><title>Human Evaluation</title><p>Fact-checking performance was evaluated by comparing assigned verdicts (&#x201C;TRUE&#x201D; or &#x201C;FALSE&#x201D;) against human assessments. Human evaluation involved verifying the correctness of each atomic verdict and categorizing errors as hallucinations, incorrect information, missing information, or lack of context. Each atomic fact correction itself was analyzed for correctness, while undetected incorrect facts were considered false negatives. Final overall answers were compared to initial responses to determine whether fact-checking led to improvement, deterioration, or no change.</p><p>Validation and tumor board analyses were conducted by one medically trained scientist (AD), supervised by JCP. Test-set evaluation used four independent blinded physician raters (AD, FB, JN, and MN), with majority voting; disagreements (2 vs 2) were resolved by a blinded fifth rater (JCP). All four raters agreed in 85% of cases, with ties in 4% of cases. The Fleiss &#x03BA; and Krippendorff &#x03B1; were both 0.16: this score is an artifact of unbalanced labels&#x2014;since the LLM classifier was highly accurate, the &#x201C;1&#x201D; (correct) label was highly prevalent, making the &#x201C;agreement expected by chance&#x201D; very high.</p><p>The following metrics were computed: true positive (verdict &#x201C;FALSE&#x201D; confirmed by a human), true negative (verdict &#x201C;TRUE&#x201D; confirmed by a human), false positive (verdict &#x201C;FALSE&#x201D; not confirmed by a human), and false negative (verdict &#x201C;TRUE&#x201D; not confirmed by a human). Standard confusion matrix metrics for classification tasks (sensitivity, specificity, precision, <italic>F</italic><sub>1</sub>, and accuracy) were calculated. Annotator guidelines are available in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-4"><title>Auto Evaluation</title><p>The framework was further evaluated on the AMEGA (Autonomous Medical Evaluation for Guideline Adherence) benchmark [<xref ref-type="bibr" rid="ref15">15</xref>], which tests adherence of LLMs to medical guidelines across 20 clinical domains. This benchmark comprises 20 patient cases (6&#x2010;8 questions each; 135 in total), with 1337 predefined evaluation criteria that can be used for automatic evaluation of generated answers. In the original study, responses were improved by iterative &#x201C;question reasking&#x201D;. Here, we applied our atomic fact-checking pipeline and compared the initial RAG answers with the corrected answers using the same criteria. The auto-evaluation was done using GPT-4o (gpt-4o-2024-11-20), and for each case, the relevant medical guidelines served as the retrieval database.</p></sec><sec id="s2-5"><title>Comparison of Open-Source Generalist vs Medical Models</title><p>Our experiments focused on differences in the fact-checking performance between medical fine-tuned, open-source LLMs and their counterpart general-purpose (ie, nonmedical) models. The compared models include Gemma 3 27B (Google Deepmind) vs MedGemma 27B (Google), Llama 3 70B (Meta AI) vs OpenBioLLM 70B (Saama AI Labs), and Qwen 3 32B (Alibaba Cloud) vs Qwen 3 Medical 32B (Alibaba Cloud). The validation set (50 prostate Q&#x0026;A pairs) was used, and a human evaluation was conducted as in previous experiments.</p><p>In addition to the quantitative fact-checking evaluation, we performed a qualitative automated evaluation. Using the LLM-as-a-judge technique, we defined 7 rubrics that evaluated different aspects of the generated answers&#x2019; quality. We used GPT-4.1 as an independent judge model, which produced a numeric score (range 0&#x2010;1) for each answer. Each rubric used finely crafted evaluation criteria and steps, which were based on common metrics from a related LLM-as-a-judge framework, G-Eval [<xref ref-type="bibr" rid="ref16">16</xref>] and adjusted by a medical expert (JCP) where needed. The rubrics used were correctness, completeness, clarity, context faithfulness, coherence, medical harmfulness, and calibration. Evaluation steps are listed in Table S8 of <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-6"><title>Baseline Approach</title><p>In order to evaluate how well our atomic fact-checking framework performs compared to other methods, we compared it with the popular framework self-refine [<xref ref-type="bibr" rid="ref17">17</xref>]. In this framework, the main idea is to first generate an initial response using an LLM, then use the same LLM to provide feedback on how to improve the response, and finally use the same LLM to refine the response based on that feedback. We used the prompts from the original paper, slightly adapted for the Q&#x0026;A use case (prompts can be found in Table S4 of <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). We evaluated the approach on the AMEGA benchmark using GPT-4o, GPT-4o-mini, Gemma 3 27B, MedGemma 27B, and Llama 3.2 3B.</p></sec><sec id="s2-7"><title>Ethical Considerations</title><p>Institutional review board approval was obtained from the Institutional Review Board of the University Hospital of the Technical University of Munich (ethics approval number 2023&#x2010;626_1 S-NP). All patients were treated after obtaining informed consent. Additional informed consent for the scientific study was not necessary due to local legislation (Bayerisches Krankenhausgesetz [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>]). All patient data were fully anonymized. There was no participant compensation.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Architectural Structure</title><p>Across all evaluation sets, answers were split into a median of 7 (IQR 5&#x2010;8.75) atomic facts, yielding a total number of 428, 404, 519, and 474 facts for the validation (50 Q&#x0026;A), testing (60 Q&#x0026;A), tumor board (40 Q&#x0026;A), and neurology (65 Q&#x0026;A) sets, respectively. Evaluation on the validation set revealed the best architectural strategy as 4-shot prompting for answer generation, verdicting, and fact rewriting (ablation study: Table S7 and Figure S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). In the atomic fact correction step, retrieving new chunks for each fact, instead of using the initial chunks (used for Q&#x0026;A) for fact-checking and rewriting, increased the overall performance (both precision and sensitivity).</p><p>Looping through all facts labeled as &#x201C;FALSE&#x201D; further increased the true positive rate and positive predictive value while reducing falsifications in rewritten facts. The average false positive rate over 3 evaluation sets decreased from 2% to 1% to 0% throughout 3 iterations of the correction loop. Therefore, 3 iterations were chosen as the maximum, since this was enough in experiments to correct all unsupported facts. Ensembling for atomic veracity prediction and changes in temperature during ensembling did not yield a significant gain in performance (Figures S2 and S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). While it slightly increased the overall <italic>F</italic><sub>1</sub>-score, it also reduced sensitivity, the most important metric in our system. Therefore, our final pipeline uses 4-shot examples and 3-step looping but no veracity prediction ensembling.</p><p>Adding few-shot examples to LLM prompts decreased the precision scores (95, 71, and 61 for 0, 1, and 4-shot settings) but increased the sensitivity (recall) scores (44, 62, and 78). Sensitivity was preferred because it is more important in our system to detect any hallucinations, while any false positives could be dismissed during the rewriting phase.</p></sec><sec id="s3-2"><title>Fact-Checking Impact in Numbers and Sets</title><p>In the validation-Q&#x0026;A and test-Q&#x0026;A sets, this final framework achieved balanced accuracy scores of 87% and 74%, with hallucination detection of 50% and 38% and inaccuracy detection of 58% and 50%, respectively (<xref ref-type="table" rid="table1">Table 1</xref>). In the more complex tumor board test set, 25% of hallucinations and 24% of inaccuracies were found, with a balanced accuracy of 72%. The positive predictive value for false fact identification was 66%, 100%, and 85% for the validation, test, and tumor board sets, respectively.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Human evaluation on 4 benchmark datasets. We present results for different metrics, as rated by medical expert annotators, across our 3 constructed and 1 external question and answer (Q&#x0026;A) datasets.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Evaluation metric</td><td align="left" valign="bottom">Validation set<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="bottom">Test set<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="bottom">Tumor board<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="bottom">Neurology Q&#x0026;A<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> [<xref ref-type="bibr" rid="ref2">2</xref>]</td></tr></thead><tbody><tr><td align="left" valign="top">Sensitivity (recall)</td><td align="left" valign="top">78</td><td align="left" valign="top">47</td><td align="left" valign="top">46</td><td align="left" valign="top">52</td></tr><tr><td align="left" valign="top">Specificity</td><td align="left" valign="top">96</td><td align="left" valign="top">100</td><td align="left" valign="top">99</td><td align="left" valign="top">100</td></tr><tr><td align="left" valign="top">Precision (PPV<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup>)</td><td align="left" valign="top">66</td><td align="left" valign="top">100</td><td align="left" valign="top">85</td><td align="left" valign="top">86</td></tr><tr><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">71</td><td align="left" valign="top">64</td><td align="left" valign="top">60</td><td align="left" valign="top">65</td></tr><tr><td align="left" valign="top">Balanced accuracy</td><td align="left" valign="top">87</td><td align="left" valign="top">74</td><td align="left" valign="top">72</td><td align="left" valign="top">76</td></tr><tr><td align="left" valign="top">TP<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup> improved atoms</td><td align="left" valign="top">100</td><td align="left" valign="top">100</td><td align="left" valign="top">97</td><td align="left" valign="top">92</td></tr><tr><td align="left" valign="top">FP<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup> falsified atoms</td><td align="left" valign="top">0</td><td align="left" valign="top">0</td><td align="left" valign="top">17</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top">Hallucination rate</td><td align="left" valign="top">1</td><td align="left" valign="top">2</td><td align="left" valign="top">1</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top">Hallucination detection</td><td align="left" valign="top">50</td><td align="left" valign="top">38</td><td align="left" valign="top">25</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top">Inaccuracy rate</td><td align="left" valign="top">3</td><td align="left" valign="top">1</td><td align="left" valign="top">5</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top">Inaccuracy detection</td><td align="left" valign="top">58</td><td align="left" valign="top">50</td><td align="left" valign="top">27</td><td align="left" valign="top">50</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Numbers represent the percentage values.</p></fn><fn id="table1fn2"><p><sup>b</sup>PPV: positive predictive value.</p></fn><fn id="table1fn3"><p><sup>c</sup>TP: true positive.</p></fn><fn id="table1fn4"><p><sup>d</sup>FP: false positive.</p></fn></table-wrap-foot></table-wrap><p>Overall answer improvements were seen in 20%, 10%, and 50% of cases in the validation, test, and tumor board&#x2013;Q&#x0026;A sets, respectively (<xref ref-type="fig" rid="figure2">Figure 2</xref>). Overall answer quality decreased in 8%, 0%, and 7.5 % of cases. The quality of the remaining answers remained the same.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Change in final answer quality. We show how the overall answer quality changed by comparing the initial and final fact-checked answer. This is shown for our 4 evaluation question and answer (Q&#x0026;A) datasets in 3 categories (improved, equal, and worse), as rated by human expert annotators. The figure displays changes by fact-checking; hence unchanged answers (no FALSE facts) are not included (validation set 56%, test set 88.3%, tumor board Q&#x0026;A 20%, neurology Q&#x0026;A 81.8%).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92090_fig02.png"/></fig><p>Achieving results comparable to our own test set on the neurology Q&#x0026;A shows that no overfitting occurred. Presenting with a sensitivity of 52%, a precision of 86%, and a balanced accuracy of 76%, the overall answer quality was improved, remained stable, or worsened in 7.7%, 7.7%, and 3% of cases, respectively (<xref ref-type="fig" rid="figure2">Figure 2</xref>).</p><p>The AMEGA auto-evaluation analysis revealed a significant improvement in answer quality by applying our fact-checking algorithm in contrast to RAG only for all LLMs tested (<italic>P</italic>&#x003C;.001 for 13/16 models, <italic>P</italic>&#x003C;.01 for 2/16 models, and <italic>P</italic>&#x003C;.05 for 1/16 models; <xref ref-type="fig" rid="figure3">Figure 3</xref>; see Table S6 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The overall best answer quality (27.3) among nonreasoning models was achieved by GPT-4o-mini and Mistral 24B, after fact-checking was applied. The largest improvement with fact-checking was seen for Llama 3.2 3B. A significant negative correlation was found between the logarithm of the LLM model parameter size (in billions) and model improvement (Pearson correlation &#x2212;0.754, <italic>P</italic>=.03; <xref ref-type="fig" rid="figure4">Figure 4</xref>). Even for the reasoning model OpenAI-o1, performance was significantly increased (<italic>P</italic>&#x003C;.001), achieving the overall best model performance of 31.52.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Results of auto-evaluation on the AMEGA (Autonomous Medical Evaluation for Guideline Adherence) benchmark. Scores for 16 different open-source and closed-source large language models (LLMs), with and without answer fact-checking. RAG: retrieval-augmented generation. ***<italic>P</italic>&#x003C;.001, **<italic>P</italic>&#x003C;.01, *<italic>P</italic>&#x003C;.05.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92090_fig03.png"/></fig><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Results of auto-evaluation on the AMEGA (Autonomous Medical Evaluation for Guideline Adherence) benchmark. Fitted linear curve of the natural logarithm of large language model (LLM) parameter size, in billions, vs the average absolute score improvement from using fact-checking for that model.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92090_fig04.png"/></fig></sec><sec id="s3-3"><title>Explainability</title><p>An important aspect of the fact-checking framework is the explainability achieved by tracing each atomic fact to the most relevant chunk (passage) in medical guidelines. To assess this<italic>,</italic> we used the test-Q&#x0026;A set and compared three similarity definitions of the best-fitting chunks selected from the vector database. A simple chain-of-thought prompt using GPT-4o identified the correct chunk in 75% of cases as the first choice and in 91.9% of cases among the top 3 chunks, outperforming a complex chain-of-thought prompt (instructing the LLM to assign scores and then rank) and cosine similarity with a text-embedding model (see Figure S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> and Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> for prompts).</p></sec><sec id="s3-4"><title>Open-Source Medical vs Nonmedical Models</title><p>In the comparison of open-source medical and nonmedical models (<xref ref-type="table" rid="table2">Table 2</xref>), MedGemma 27B performed best overall, with a balanced accuracy of 90%, a sensitivity of 83%, and a total improvement in answers of 30%, outperforming even the GPT-4o baseline. The high sensitivity score of MedGemma 27B was better than that of its Gemma 3 27B counterpart and GPT-4o (83% vs 46% vs 78%).</p><p>The 2 tested Qwen models showed comparable sensitivity (55% vs 56%), while Llama 3 70B considerably outperformed OpenBioLLM 70B (61% vs 33%). The same tendency holds true for their <italic>F</italic><sub>1</sub>-scores and balanced accuracies. The positive predictive value was the highest for OpenBioLLM 70B; however, it had markedly reduced sensitivity and an elevated false-negative rate.</p><p>Looking at the improved overall answer quality as the main fact-checking outcome, MedGemma 27B demonstrated superior performance compared to its generalist counterpart, Gemma 3 27B, as well as the GPT-4o baseline (30% vs 20% vs 20%) and all other evaluated models. The Qwen models showed moderate results (generalist: 18% and medical: 24%), while Llama 3 surpassed OpenBioLLM (8% vs 4%). Among nonmedical models, Llama 3 achieved the highest balanced accuracy (79%).</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Comparing the fact-checking performance of multiple open-source large language models (LLMs) and the proprietary GPT-4o (the baseline used in previous experiments)<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Evaluation metric<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td><td align="left" valign="bottom">GPT-4o (baseline)</td><td align="left" valign="bottom">Gemma 3 27B</td><td align="left" valign="bottom">MedGemma 27B</td><td align="left" valign="bottom">Llama 3 70B</td><td align="left" valign="bottom">OpenBioLLM 70B</td><td align="left" valign="bottom">Qwen 3 32B</td><td align="left" valign="bottom">Qwen 3 Medical 32B</td></tr></thead><tbody><tr><td align="left" valign="top">TP<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td><td align="left" valign="top">6</td><td align="left" valign="top">6</td><td align="left" valign="top">11</td><td align="left" valign="top">12</td><td align="left" valign="top">7</td><td align="left" valign="top">8</td><td align="left" valign="top">8</td></tr><tr><td align="left" valign="top">FP<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="top">3</td><td align="left" valign="top">3</td><td align="left" valign="top">3</td><td align="left" valign="top">2</td><td align="left" valign="top">1</td><td align="left" valign="top">2</td><td align="left" valign="top">3</td></tr><tr><td align="left" valign="top">TN<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup></td><td align="left" valign="top">89</td><td align="left" valign="top">84</td><td align="left" valign="top">84</td><td align="left" valign="top">78</td><td align="left" valign="top">79</td><td align="left" valign="top">83</td><td align="left" valign="top">82</td></tr><tr><td align="left" valign="top">FN<sup><xref ref-type="table-fn" rid="table2fn6">f</xref></sup></td><td align="left" valign="top">2</td><td align="left" valign="top">7</td><td align="left" valign="top">2</td><td align="left" valign="top">8</td><td align="left" valign="top">14</td><td align="left" valign="top">6</td><td align="left" valign="top">7</td></tr><tr><td align="left" valign="top">Sensitivity (recall)</td><td align="left" valign="top">78</td><td align="left" valign="top">46</td><td align="left" valign="top">83</td><td align="left" valign="top">61</td><td align="left" valign="top">33</td><td align="left" valign="top">56</td><td align="left" valign="top">55</td></tr><tr><td align="left" valign="top">Specificity</td><td align="left" valign="top">96</td><td align="left" valign="top">97</td><td align="left" valign="top">97</td><td align="left" valign="top">97</td><td align="left" valign="top">99</td><td align="left" valign="top">98</td><td align="left" valign="top">96</td></tr><tr><td align="left" valign="top">Precision (PPV<sup><xref ref-type="table-fn" rid="table2fn7">g</xref></sup>)</td><td align="left" valign="top">66</td><td align="left" valign="top">68</td><td align="left" valign="top">79</td><td align="left" valign="top">83</td><td align="left" valign="top">88</td><td align="left" valign="top">80</td><td align="left" valign="top">72</td></tr><tr><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">71</td><td align="left" valign="top">55</td><td align="left" valign="top">81</td><td align="left" valign="top">70</td><td align="left" valign="top">48</td><td align="left" valign="top">66</td><td align="left" valign="top">63</td></tr><tr><td align="left" valign="top">Accuracy</td><td align="left" valign="top">95</td><td align="left" valign="top">90</td><td align="left" valign="top">95</td><td align="left" valign="top">90</td><td align="left" valign="top">86</td><td align="left" valign="top">92</td><td align="left" valign="top">90</td></tr><tr><td align="left" valign="top">Balanced accuracy</td><td align="left" valign="top">87.3</td><td align="left" valign="top">71</td><td align="left" valign="top">90</td><td align="left" valign="top">79</td><td align="left" valign="top">66</td><td align="left" valign="top">77</td><td align="left" valign="top">76</td></tr><tr><td align="left" valign="top">TP improved atoms</td><td align="left" valign="top">100</td><td align="left" valign="top">100</td><td align="left" valign="top">91</td><td align="left" valign="top">98</td><td align="left" valign="top">100</td><td align="left" valign="top">82</td><td align="left" valign="top">89</td></tr><tr><td align="left" valign="top">FP falsified atoms</td><td align="left" valign="top">0</td><td align="left" valign="top">30</td><td align="left" valign="top">44</td><td align="left" valign="top">50</td><td align="left" valign="top">0</td><td align="left" valign="top">0</td><td align="left" valign="top">14</td></tr><tr><td align="left" valign="top">Overall answer quality improved</td><td align="left" valign="top">20</td><td align="left" valign="top">20</td><td align="left" valign="top">30</td><td align="left" valign="top">8</td><td align="left" valign="top">4</td><td align="left" valign="top">18</td><td align="left" valign="top">24</td></tr><tr><td align="left" valign="top">Overall answer quality equal</td><td align="left" valign="top">14</td><td align="left" valign="top">8</td><td align="left" valign="top">6</td><td align="left" valign="top">4</td><td align="left" valign="top">4</td><td align="left" valign="top">8</td><td align="left" valign="top">22</td></tr><tr><td align="left" valign="top">Overall answer quality worse</td><td align="left" valign="top">8</td><td align="left" valign="top">4</td><td align="left" valign="top">6</td><td align="left" valign="top">6</td><td align="left" valign="top">10</td><td align="left" valign="top">16</td><td align="left" valign="top">10</td></tr><tr><td align="left" valign="top">Hallucination rate</td><td align="left" valign="top">1</td><td align="left" valign="top">2</td><td align="left" valign="top">3</td><td align="left" valign="top">6</td><td align="left" valign="top">11</td><td align="left" valign="top">4</td><td align="left" valign="top">2</td></tr><tr><td align="left" valign="top">Hallucination detection</td><td align="left" valign="top">50</td><td align="left" valign="top">71</td><td align="left" valign="top">80</td><td align="left" valign="top">52</td><td align="left" valign="top">50</td><td align="left" valign="top">69</td><td align="left" valign="top">60</td></tr><tr><td align="left" valign="top">Inaccuracy rate</td><td align="left" valign="top">3</td><td align="left" valign="top">2</td><td align="left" valign="top">3</td><td align="left" valign="top">5</td><td align="left" valign="top">3</td><td align="left" valign="top">3</td><td align="left" valign="top">4</td></tr><tr><td align="left" valign="top">Inaccuracy detection</td><td align="left" valign="top">58</td><td align="left" valign="top">38</td><td align="left" valign="top">90</td><td align="left" valign="top">68</td><td align="left" valign="top">22</td><td align="left" valign="top">21</td><td align="left" valign="top">44</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>The focus was on general-purpose vs medical fine-tuned models of different sizes: Gemma 3 27B vs MedGemma 27B; Llama 3 70B vs OpenBioLLM 70B; and Qwen 3 32B vs Qwen 3 Medical 32B.</p></fn><fn id="table2fn2"><p><sup>b</sup>Numbers represent the percentage values.</p></fn><fn id="table2fn3"><p><sup>c</sup>TP: true positive.</p></fn><fn id="table2fn4"><p><sup>d</sup>FP: false positive.</p></fn><fn id="table2fn5"><p><sup>e</sup>TN: true negative.</p></fn><fn id="table2fn6"><p><sup>f</sup>FN: false negative.</p></fn><fn id="table2fn7"><p><sup>g</sup>PPV: positive predictive value.</p></fn></table-wrap-foot></table-wrap><p>Results of LLM-as-a-judge automated rubric evaluation are shown for the baseline, GPT-4o, and the best-performing open-source model based on <italic>F</italic><sub>1</sub>-score, MedGemma 27B, in <xref ref-type="table" rid="table3">Table 3</xref>. Results of all open-source models are available in Table S9 of <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Even though MedGemma 27B outperformed GPT-4o on the fact-checking task (atomic claim veracity prediction), GPT-4o scored better in four rubrics, MedGemma in one rubric, while scores were almost equal in two rubrics. While MedGemma improved more answers in total (30% vs 20%), certain aspects of its answers are still lacking compared to GPT. Most of MedGemma&#x2019;s answers had significantly higher rubric scores after the fact-checking process was performed. When evaluating all LLMs together, context faithfulness was significantly improved (Table S9 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Comparison of large language model (LLM)&#x2013;as-a-judge rubric scores<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">LLM or metric</td><td align="left" valign="bottom" colspan="2">Correctness,<break/>mean (95% CI)</td><td align="left" valign="bottom" colspan="2">Completeness,<break/>mean (95% CI)</td><td align="left" valign="bottom" colspan="2">Clarity,<break/>mean (95% CI)</td><td align="left" valign="bottom" colspan="2">Context faithfulness,<break/>mean (95% CI)</td><td align="left" valign="bottom" colspan="2">Coherence,<break/>mean (95% CI)</td><td align="left" valign="bottom" colspan="2">Medical harmfulness,<break/>mean (95% CI)</td><td align="left" valign="bottom" colspan="2">Calibration,<break/>mean (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">Baseline or final<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td><td align="left" valign="top">Baseline</td><td align="left" valign="top">Final</td><td align="left" valign="top">Baseline</td><td align="left" valign="top">Final</td><td align="left" valign="top">Baseline</td><td align="left" valign="top">Final</td><td align="left" valign="top">Baseline</td><td align="left" valign="top">Final</td><td align="left" valign="top">Baseline</td><td align="left" valign="top">Final</td><td align="left" valign="top">Baseline</td><td align="left" valign="top">Final</td><td align="left" valign="top">Baseline</td><td align="left" valign="top">Final</td></tr><tr><td align="left" valign="top">GPT-4o</td><td align="left" valign="top">66<break/>(60&#x2010;72)</td><td align="left" valign="top">66<break/>(60&#x2010;73)</td><td align="left" valign="top">93<break/>(91&#x2010;95)</td><td align="left" valign="top">92<break/>(90-94)</td><td align="left" valign="top">87<break/>(86&#x2010;88)</td><td align="left" valign="top">86<break/>(85&#x2010;87)</td><td align="left" valign="top">88<break/>(85&#x2010;91)</td><td align="left" valign="top">88<break/>(84&#x2010;91)</td><td align="left" valign="top">89<break/>(89&#x2010;90)</td><td align="left" valign="top">89<break/>(89-90)</td><td align="left" valign="top">95<break/>(92&#x2010;98)</td><td align="left" valign="top">94<break/>(91&#x2010;97)</td><td align="left" valign="top">87<break/>(86&#x2010;89)</td><td align="left" valign="top">87<break/>(85&#x2010;89)</td></tr><tr><td align="left" valign="top">MedGemma 27B</td><td align="left" valign="top">57<break/>(49&#x2010;65)</td><td align="left" valign="top">58<break/>(50-65)</td><td align="left" valign="top">81<break/>(76&#x2010;87)</td><td align="left" valign="top">86<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup><break/>(82-91)</td><td align="left" valign="top">80<break/>(78&#x2010;82)</td><td align="left" valign="top">83<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup><break/>(81-85)</td><td align="left" valign="top">90<break/>(87&#x2010;92)</td><td align="left" valign="top">92<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup><break/>(91-93)</td><td align="left" valign="top">76<break/>(71&#x2010;81)</td><td align="left" valign="top">80<break/>(76-83)</td><td align="left" valign="top">94<break/>(92&#x2010;95)</td><td align="left" valign="top">95<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup><break/>(93-97)</td><td align="left" valign="top">86<break/>(84&#x2010;87)</td><td align="left" valign="top">87<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup><break/>(85-89)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>We show the final auto-evaluation scores for 7 rubrics, comparing the baseline model, GPT-4o, and the best-performing open-source model, MedGemma 27B. The scores are averaged across 50 answers on the validation set.</p></fn><fn id="table3fn2"><p><sup>b</sup>Baseline scores are for initial answers based on retrieval-augmented generation (RAG). Final scores are for final answers after fact-checking was performed.</p></fn><fn id="table3fn3"><p><sup>c</sup><italic>P</italic>&#x003C;.05.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-5"><title>Baseline Comparison</title><p>Comparison with the competing self-correcting framework self-refine [<xref ref-type="bibr" rid="ref17">17</xref>] on AMEGA is shown in <xref ref-type="table" rid="table4">Table 4</xref>. The improvement rate using self-refine depended on the model evaluated. The highest rate was seen for GPT-4o, while Gemini models and open-source models showed lower performance improvements. In comparison with self-refine, fact-checking was significantly better for two smaller open-source models, while there was no difference for recent Gemini models. For GPT-4o, self-refine showed significantly better performance.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Comparison of our fact-checking and the self-correcting self-refine approaches using AMEGA (Autonomous Medical Evaluation for Guideline Adherence)<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup>.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Initial score on AMEGA</td><td align="left" valign="bottom">With fact-checking (our approach)</td><td align="left" valign="bottom">With self-refine [<xref ref-type="bibr" rid="ref17">17</xref>]</td><td align="left" valign="bottom">Difference fact-checking to self-refine</td></tr></thead><tbody><tr><td align="left" valign="top">GPT-4o</td><td align="left" valign="top">25.4</td><td align="left" valign="top">26.5 (+1.1)<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">28.3 (+2.9)<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">&#x2013;1.8<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td></tr><tr><td align="left" valign="top">GPT-4o-mini</td><td align="left" valign="top">26.3</td><td align="left" valign="top">27.3 (+1.0)<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">28.0 (+1.7)<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">&#x2013;0.7</td></tr><tr><td align="left" valign="top">Gemini 3.1 Pro</td><td align="left" valign="top">16.0</td><td align="left" valign="top">17.6 (+1.6)<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">17.5 (+1.5)<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">+0.1</td></tr><tr><td align="left" valign="top">Gemini 3.1 Flash Lite</td><td align="left" valign="top">17.6</td><td align="left" valign="top">19.7 (+2.1)<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">19.1 (+1.5)<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">+0.6</td></tr><tr><td align="left" valign="top">Gemini 3.5 Flash</td><td align="left" valign="top">19.3</td><td align="left" valign="top">20.5 (+1.2)<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">20.0 (+0.7)<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">+0.5</td></tr><tr><td align="left" valign="top">Gemma 3 27B</td><td align="left" valign="top">18.1</td><td align="left" valign="top">20.0 (+1.9)<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">19.4 (+1.3)<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">+0.6</td></tr><tr><td align="left" valign="top">MedGemma 27B</td><td align="left" valign="top">21.8</td><td align="left" valign="top">24.2 (+2.4)<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">22.8 (+1.0)<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">+1.4<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td></tr><tr><td align="left" valign="top">Llama 3.2 3B</td><td align="left" valign="top">20.2</td><td align="left" valign="top">24.6 (+4.4)<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">21.0 (+0.8)<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">+3.6<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>We show the final auto-evaluation scores on AMEGA for five representative large language models (LLMs) based on (1) initial responses; (2) responses corrected with our fact-checking framework; and (3) responses corrected with the competing self-refine approach.</p></fn><fn id="table4fn2"><p><sup>b</sup><italic>P</italic>&#x003C;.001, compared with the initial score.</p></fn><fn id="table4fn3"><p><sup>c</sup><italic>P</italic>&#x003C;.05.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><p>Our study introduces the first application of an atomic fact-checking framework designed to enhance the reliability and explainability of LLMs used in medical Q&#x0026;A. By decomposing LLM-generated responses into discrete atomic facts and rigorously verifying each against an authoritative vector database, the framework significantly reduced hallucinations and inaccuracies. Medical expert assessment and automated benchmarks demonstrated notable improvements in factual accuracy, achieving up to a 50% overall answer improvement and an 80% hallucination detection rate. Additionally, the framework achieved high explainability by tracing each atomic fact back to the most relevant chunks from the database, providing a granular, transparent explanation of the generated responses. Rubric-based autoevaluation not only revealed an LLM dependency in score improvement but also significantly increased source faithfulness across all LLMs. Compared with the self-refine approach, performance gains differed depending on the model chosen. While the former GPT-4o frontier model profited from self-refinement, there was no difference for current Gemini models, and smaller open-source models significantly benefited from atomic fact-checking.</p><p>Our framework increased the factual accuracy and overall quality of LLM-generated responses. Numerical hallucinations, such as incorrect drug dosages, were frequently identified and corrected, as were entity hallucinations, such as the conflation of different treatment procedures. In a small proportion of cases (8%, 0%, 7.5%, and 3% across Q&#x0026;A sets), the answer quality declined due to the retrieval of less relevant chunks or to potential hallucinations introduced during the correction process. However, since the proportion of answers that improved (20%, 10%, 50%, and 7.7%) was substantially higher, the overall effect of fact-checking clearly enhanced the answer quality. Different degrees of improvement come down to the varying complexity of questions across the datasets and to different rates of atomic facts being identified as FALSE and corrected.</p><p>Notably, the observed gain was strongest in real tumor-board questions. The most challenging and clinically realistic dataset achieved the highest rate of answer improvement (50%). The other 3 Q&#x0026;A datasets (validation, test, and neurology-Q&#x0026;A sets), where questions were straightforward and usually answerable directly from guideline passages, resulted in moderate improvement after fact-checking (20%, 10%, and 7.7%). Owing to this close correspondence, the potential for further improvement through fact-checking is inherently limited. In contrast, the tumor board cases were substantially more complex and based on real-world patient scenarios rather than guideline excerpts. Consequently, there was no exact blueprint answer available in the source material, leaving greater room for iterative refinement and revision of responses. This suggests that the system contributes most when queries are complex and multifactual, for example, in clinical settings.</p><p>Regarding the performance of medical fine-tuned LLMs in detecting incorrect facts, the model size appeared to be the most influential factor. As smaller models are less capable of answering complex questions, they derive great benefit from post hoc fact-checking (<xref ref-type="fig" rid="figure4">Figure 4</xref>). Fine-tuning for the medical domain can boost task specialization, but the effect is strongly dependent on individual details, differing from model to model and from adaptation to adaptation, and making direct comparison difficult. We observed that medical fine-tuned models mostly had worse performance, with MedGemma 27B marking an exception. Hence, it cannot be said that medical fine-tuned models show superior performance and improvement in general.</p><p>MedGemma 27B achieved notable results and improvements through fact-checking, outperforming even the GPT-4o baseline. Being a reasoning model (it outputs &#x201C;thinking&#x201D; tokens before the final response) likely helps its performance. Reasoning models also have the additional benefit of providing more interpretable answers. Additionally, MedGemma may have benefited from more effective fine-tuning compared with the other models.</p><p>Consistently, the answer quality of MedGemma 27B&#x2019;s answers improved according to LLM-as-a-judge evaluation metrics after fact-checking was applied. Nevertheless, its performance remained below that of the proprietary GPT-4o model, indicating room for improvement.</p><p>RAG-based medical chatbots may qualify as Class IIa medical devices according to the European Medical Device Regulation (MDR) [<xref ref-type="bibr" rid="ref3">3</xref>]. Trustworthiness, the combination of explainability, traceability, and transparency,&#x202F;is a key prerequisite under the MDR [<xref ref-type="bibr" rid="ref20">20</xref>]. Although LLMs inherently explain their responses, these justifications can be misleading. Pure RAG remains susceptible to intrinsic limitations of LLMs, such as the incorporation of unsupported internal knowledge or incorrect synthesis of information across retrieved chunks. Thus, retrieval augmentation alone cannot fully guarantee factual consistency [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref22">22</xref>]. Atomic fact-checking serves as an explicit verification layer, systematically validating the generated claims, thereby improving factual reliability beyond what RAG alone can achieve [<xref ref-type="bibr" rid="ref23">23</xref>]. LLMs improving over time might alleviate these issues, but, as they will always be stochastic, a rigorous verification mechanism remains necessary, especially in safety-critical domains such as medicine.</p><p>The fact-checking framework provides an additional advantage beyond answer refinement alone: it enables fact-wise verification and traceability of the generated statements back to the supporting source documents. This fine level of explainability and evidence grounding is another strength of the framework. Importantly, this process is outsourced from the internal reasoning of LLMs and directed toward a potential user. While our current work does not constitute a complete analysis of trustworthiness in all its dimensions, findings from a randomized controlled setting [<xref ref-type="bibr" rid="ref24">24</xref>] provide relevant evidence: presenting AI-generated recommendations decomposed into individually verifiable claims and linked to source guidelines, thereby presenting a practical application of the fact-checking algorithm, was associated with substantially higher clinician trust than traditional explainability approaches.</p><p>While the self-refine method achieved similar improvements in answer quality in larger models (greater for GPT models but lower for Gemini models), we demonstrated that the benefit of our framework was greater in smaller LLMs. Smaller models achieved higher improvements using the fact-checking framework on benchmarks, including AMEGA, where our approach also outperformed the competing single-pass correction baseline. The even higher usability of the fact-checking algorithm in smaller LLMs is especially interesting, considering scenarios of potential on-premises deployment in medical institutions. The fact-checking approach offers explainability advantages that self-refine lacks. The generally lower Gemini scores may stem from prompts being optimized for the GPT model family and, for consistency, were used for all experiments.</p><p>Our work has several limitations. As we sought to evaluate open Q&#x0026;A capabilities, we designed novel evaluation datasets. Our evaluation included 215 human-evaluated Q&#x0026;As that were supplemented by autoevaluations using the AMEGA benchmark with 1337 scoring elements. Altogether, this corresponds to 1552 evaluated question-answer instances used in this study. This number is limited, and larger scale validation could further strengthen generalizability; however, the size of our manually evaluated datasets is comparable to that of many prior studies in the medical AI domain, where human expert annotation and review require substantial time and resources. Nevertheless, evaluation on a large-scale public benchmark is an important future step. Furthermore, all pipeline steps rely on LLM generation. It is possible that errors can propagate from one step to another. Future work could explore approaches to make the process more rigorous, such as Graph-RAG techniques for grounding data into graph structures for more robust generation. Finally, the entire fact-checking and rewriting process uses around 10 times more tokens than just the initial response generation (as shown in Table S11 of the <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), which increases the cost and the latency of the system. A more token-efficient approach for our application is currently under investigation.</p><p>But still, in a world of rapidly generated and propagated data, it is more important than ever to check your facts thoroughly.</p><p>To conclude, we present the application of an atomic fact-checking algorithm that identifies factual inaccuracies and hallucinations in medical Q&#x0026;A. Correcting these findings improves the overall answer quality while achieving fact-wise explainability, paving the way for more credible clinical use of LLMs.</p></sec></body><back><ack><p>No generative AI has been used in any portion of the manuscript generation.</p></ack><notes><sec><title>Funding</title><p>JCP and FM received funding from the Google.org Accelerator for Generative AI (2025). The funder played no role in study design, data collection, analysis and interpretation of data, or the writing of this manuscript.</p></sec><sec><title>Data Availability</title><p>The datasets and code used to perform the analyses in this study are available publicly on GitHub [<xref ref-type="bibr" rid="ref22">22</xref>].</p></sec></notes><fn-group><fn fn-type="con"><p>JV and AD conducted the experiments, evaluation, and results analysis, and wrote the initial manuscript. FM and JCP provided supervision and guidance. MN, RM, JN, FB, LA, KKB, DB, SEC, and KB contributed to data curation and interpretation, discussed the results, and edited the manuscript.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AMEGA</term><def><p>Autonomous Medical Evaluation for Guideline Adherence</p></def></def-item><def-item><term id="abb2">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb3">MDR</term><def><p>Medical Device Regulation</p></def></def-item><def-item><term id="abb4">Q&#x0026;A</term><def><p>question and answer</p></def></def-item><def-item><term id="abb5">RAG</term><def><p>retrieval-augmented generation</p></def></def-item><def-item><term id="abb6">TRIPOD</term><def><p>Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Han</surname><given-names>T</given-names> </name><name name-style="western"><surname>Adams</surname><given-names>LC</given-names> </name><name name-style="western"><surname>Bressem</surname><given-names>KK</given-names> </name><name name-style="western"><surname>Busch</surname><given-names>F</given-names> </name><name name-style="western"><surname>Nebelung</surname><given-names>S</given-names> </name><name name-style="western"><surname>Truhn</surname><given-names>D</given-names> </name></person-group><article-title>Comparative analysis of multimodal large language model performance on clinical vignette questions</article-title><source>JAMA</source><year>2024</year><month>04</month><day>16</day><volume>331</volume><issue>15</issue><fpage>1320</fpage><lpage>1321</lpage><pub-id pub-id-type="doi">10.1001/jama.2023.27861</pub-id><pub-id pub-id-type="medline">38497956</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Masanneck</surname><given-names>L</given-names> </name><name name-style="western"><surname>Meuth</surname><given-names>SG</given-names> </name><name name-style="western"><surname>Pawlitzki</surname><given-names>M</given-names> </name></person-group><article-title>Evaluating base and retrieval augmented LLMs with document or online support for evidence based neurology</article-title><source>NPJ Digit Med</source><year>2025</year><month>03</month><day>4</day><volume>8</volume><issue>1</issue><fpage>137</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01536-y</pub-id><pub-id pub-id-type="medline">40038423</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Freyer</surname><given-names>O</given-names> </name><name name-style="western"><surname>Wiest</surname><given-names>IC</given-names> </name><name name-style="western"><surname>Kather</surname><given-names>JN</given-names> </name><name name-style="western"><surname>Gilbert</surname><given-names>S</given-names> </name></person-group><article-title>A future role for health applications of large language models depends on regulators enforcing safety standards</article-title><source>Lancet Digit Health</source><year>2024</year><month>09</month><volume>6</volume><issue>9</issue><fpage>e662</fpage><lpage>e672</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(24)00124-9</pub-id><pub-id pub-id-type="medline">39179311</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ng</surname><given-names>KKY</given-names> </name><name name-style="western"><surname>Matsuba</surname><given-names>I</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>PC</given-names> </name></person-group><article-title>RAG in health care: a novel framework for improving communication and decision-making by addressing LLM limitations</article-title><source>NEJM AI</source><year>2025</year><month>01</month><volume>2</volume><issue>1</issue><pub-id pub-id-type="doi">10.1056/AIra2400380</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ferber</surname><given-names>D</given-names> </name><name name-style="western"><surname>Wiest</surname><given-names>IC</given-names> </name><name name-style="western"><surname>W&#x00F6;lflein</surname><given-names>G</given-names> </name><etal/></person-group><article-title>GPT-4 for information retrieval and comparison of medical oncology guidelines</article-title><source>NEJM AI</source><year>2024</year><month>05</month><day>23</day><volume>1</volume><issue>6</issue><pub-id pub-id-type="doi">10.1056/AIcs2300235</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kotonya</surname><given-names>N</given-names> </name><name name-style="western"><surname>Toni</surname><given-names>F</given-names> </name></person-group><article-title>Explainable automated fact-checking: a survey</article-title><source>Proc 28th Int Conf Comput Linguist</source><year>2020</year><fpage>5430</fpage><lpage>5443</lpage><pub-id pub-id-type="doi">10.18653/v1/2020.coling-main.474</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mesinovic</surname><given-names>M</given-names> </name><name name-style="western"><surname>Watkinson</surname><given-names>P</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>T</given-names> </name></person-group><article-title>Explainability in the age of large language models for healthcare</article-title><source>Commun Eng</source><year>2025</year><month>07</month><day>17</day><volume>4</volume><issue>1</issue><fpage>128</fpage><pub-id pub-id-type="doi">10.1038/s44172-025-00453-y</pub-id><pub-id pub-id-type="medline">40676176</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kamoi</surname><given-names>R</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>N</given-names> </name><name name-style="western"><surname>Han</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>R</given-names> </name></person-group><article-title>When can LLMs actually correct their own mistakes? A critical survey of self-correction of LLMs</article-title><source>Trans Assoc Comput Linguist</source><year>2024</year><month>11</month><day>4</day><volume>12</volume><fpage>1417</fpage><lpage>1440</lpage><pub-id pub-id-type="doi">10.1162/tacl_a_00713</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Augenstein</surname><given-names>I</given-names> </name><name name-style="western"><surname>Baldwin</surname><given-names>T</given-names> </name><name name-style="western"><surname>Cha</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Factuality challenges in the era of large language models and opportunities for fact-checking</article-title><source>Nat Mach Intell</source><year>2024</year><volume>6</volume><issue>8</issue><fpage>852</fpage><lpage>863</lpage><pub-id pub-id-type="doi">10.1038/s42256-024-00881-z</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Min</surname><given-names>S</given-names> </name><name name-style="western"><surname>Krishna</surname><given-names>K</given-names> </name><name name-style="western"><surname>Lyu</surname><given-names>X</given-names> </name><etal/></person-group><article-title>FActScore: fine-grained atomic evaluation of factual precision in long form text generation</article-title><source>Proc 2023 Conf Empir Methods Nat Lang Process</source><year>2023</year><fpage>12076</fpage><lpage>12100</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.emnlp-main.741</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vladika</surname><given-names>J</given-names> </name><name name-style="western"><surname>Matthes</surname><given-names>F</given-names> </name></person-group><article-title>Scientific fact-checking: a survey of resources and approaches</article-title><source>Findings Assoc Comput Linguist</source><year>2023</year><fpage>6215</fpage><lpage>6230</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.findings-acl.387</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>G</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Closing the gap between open source and commercial large language models for medical evidence summarization</article-title><source>NPJ Digit Med</source><year>2024</year><month>09</month><day>9</day><volume>7</volume><issue>1</issue><fpage>239</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01239-w</pub-id><pub-id pub-id-type="medline">39251804</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Deka</surname><given-names>P</given-names> </name><name name-style="western"><surname>Jurek-Loughrey</surname><given-names>A</given-names> </name><name name-style="western"><surname>P.</surname><given-names>D</given-names> </name></person-group><article-title>Improved methods to aid unsupervised evidence-based fact checking for online health news</article-title><source>J Data Intell</source><year>2022</year><volume>3</volume><issue>4</issue><fpage>474</fpage><lpage>504</lpage><pub-id pub-id-type="doi">10.26421/JDI3.4-5</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="web"><article-title>sebischair/ImprovingReliabilityMedicalQA</article-title><source>GitHub</source><access-date>2026-08-26</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/sebischair/ImprovingReliabilityMedicalQA">https://github.com/sebischair/ImprovingReliabilityMedicalQA</ext-link></comment></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fast</surname><given-names>D</given-names> </name><name name-style="western"><surname>Adams</surname><given-names>LC</given-names> </name><name name-style="western"><surname>Busch</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Autonomous medical evaluation for guideline adherence of large language models</article-title><source>NPJ Digit Med</source><year>2024</year><month>12</month><day>12</day><volume>7</volume><issue>1</issue><fpage>358</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01356-6</pub-id><pub-id pub-id-type="medline">39668168</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Iter</surname><given-names>D</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>R</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>C</given-names> </name></person-group><article-title>G-eval: NLG evaluation using GPT-4 with better human alignment</article-title><source>Proc Conf Empir Methods Nat Lang Process</source><year>2024</year><fpage>2511</fpage><lpage>2522</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.emnlp-main.153</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Madaan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Tandon</surname><given-names>N</given-names> </name><name name-style="western"><surname>Gupta</surname><given-names>P</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Oh</surname><given-names>A</given-names> </name><name name-style="western"><surname>Naumann</surname><given-names>T</given-names> </name><name name-style="western"><surname>Globerson</surname><given-names>A</given-names> </name><name name-style="western"><surname>Saenko</surname><given-names>K</given-names> </name><name name-style="western"><surname>Hardt</surname><given-names>M</given-names> </name><name name-style="western"><surname>Levine</surname><given-names>S</given-names> </name></person-group><article-title>SELF-REFINE: iterative refinement with SELF-feedback</article-title><source>NIPS &#x2019;23: Proceedings of the 37th International Conference on Neural Information Processing Systems</source><year>2023</year><access-date>2026-08-26</access-date><publisher-name>Curran Associates Inc</publisher-name><fpage>46534</fpage><lpage>46594</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/10.5555/3666122.3668141">https://dl.acm.org/doi/10.5555/3666122.3668141</ext-link></comment></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="web"><article-title>Bayerisches Krankenhausgesetz (BayKrG): art 27 Datenschutz [Article in German]</article-title><source>BAYERN.RECHT</source><year>2007</year><month>03</month><day>28</day><access-date>2026-09-07</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.gesetze-bayern.de/Content/Document/BayKrG-27">https://www.gesetze-bayern.de/Content/Document/BayKrG-27</ext-link></comment></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="web"><article-title>Gesetz &#x00FC;ber die Universit&#x00E4;tsklinika des Freistaates Bayern (Bayerisches Universit&#x00E4;tsklinikagesetz &#x2013; BayUniKlinG) Art 16Anwendung hochschul- und krankenhausrechtlicher Vorschriften [Article in German]</article-title><source>BAYERN.RECHT</source><year>2006</year><month>05</month><day>23</day><access-date>2026-09-07</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.gesetze-bayern.de/Content/Document/BayUniKlinG-16">https://www.gesetze-bayern.de/Content/Document/BayUniKlinG-16</ext-link></comment></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>B</given-names> </name><name name-style="western"><surname>Qi</surname><given-names>P</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Trustworthy AI: from principles to practices</article-title><source>ACM Comput Surv</source><year>2023</year><month>09</month><day>30</day><volume>55</volume><issue>9</issue><fpage>1</fpage><lpage>46</lpage><pub-id pub-id-type="doi">10.1145/3555803</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Barnett</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kurniawan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Thudumu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Brannelly</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Abdelrazek</surname><given-names>M</given-names> </name></person-group><article-title>Seven failure points when engineering a retrieval augmented generation system</article-title><source>CAIN 2024</source><year>2024</year><fpage>194</fpage><lpage>199</lpage><pub-id pub-id-type="doi">10.1145/3644815.3644945</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Xiang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Xiao</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>FaithfulRAG: fact-level conflict modeling for context-faithful retrieval-augmented generation</article-title><source>Proc Annu Meet Assoc Comput Linguist</source><year>2025</year><fpage>21863</fpage><lpage>21882</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.acl-long.1062</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rahman</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Islam</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Alam</surname><given-names>MM</given-names> </name><etal/></person-group><article-title>Hallucination to truth: a review of fact-checking and factuality evaluation in large language models</article-title><source>Artif Intell Rev</source><year>2026</year><volume>59</volume><issue>2</issue><fpage>70</fpage><pub-id pub-id-type="doi">10.1007/s10462-025-11454-w</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Adams</surname><given-names>LC</given-names> </name><name name-style="western"><surname>Marx</surname><given-names>L</given-names> </name><name name-style="western"><surname>Orberg</surname><given-names>ET</given-names> </name><etal/></person-group><article-title>Atomic fact-checking increases clinician trust in large language model recommendations for oncology decision support: a randomized controlled trial</article-title><source>arXiv</source><comment>Preprint posted online on  May 5, 2026</comment><pub-id pub-id-type="doi">10.48550/ARXIV.2605.03916</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Prompts, few-shot examples, and further experiments for atomic fact-checking in medical retrieval-augmented generation systems.</p><media xlink:href="jmir_v28i1e92090_app1.pdf" xlink:title="PDF File, 1754 KB"/></supplementary-material><supplementary-material id="app2"><label>Checklist 1</label><p>TRIPOD-LLM checklist.</p><media xlink:href="jmir_v28i1e92090_app2.pdf" xlink:title="PDF File, 304 KB"/></supplementary-material></app-group></back></article>