<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e84086</article-id><article-id pub-id-type="doi">10.2196/84086</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Intelligent Framework for Adverse Drug Event Identification Using Large Language Models and Retrieval-Augmented Generation: Development and Evaluation Study</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Ma</surname><given-names>Junlong</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Wu</surname><given-names>Xuehong</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Feng</surname><given-names>Zeying</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kuang</surname><given-names>Yun</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ding</surname><given-names>Zhendong</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Li</surname><given-names>Min</given-names></name><degrees>Prof Dr</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Yang</surname><given-names>Guoping</given-names></name><degrees>Prof Dr</degrees><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="aff" rid="aff6">6</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Pharmacy, Xiangya Hospital, Central South University</institution><addr-line>Changsha</addr-line><addr-line>Hunan</addr-line><country>China</country></aff><aff id="aff2"><institution>School of Computer Science and Engineering, Central South University</institution><addr-line>Changsha</addr-line><addr-line>Hunan</addr-line><country>China</country></aff><aff id="aff3"><institution>Clinical Trial Institution Office, Liuzhou Hospital of Guangzhou Women and Children's Medical Center</institution><addr-line>Liuzhou</addr-line><addr-line>Guangxi</addr-line><country>China</country></aff><aff id="aff4"><institution>Center of Clinical Pharmacology, The Third Xiangya Hospital, Central South University</institution><addr-line>No 138 Tongzipo Road, Yuelu District,</addr-line><addr-line>Changsha</addr-line><addr-line>Hunan</addr-line><country>China</country></aff><aff id="aff5"><institution>Department of Anesthesiology, The Third Xiangya Hospital, Central South University</institution><addr-line>Changsha</addr-line><addr-line>Hunan</addr-line><country>China</country></aff><aff id="aff6"><institution>Xiangya School of Pharmaceutical Sciences, Central South University</institution><addr-line>Changsha</addr-line><addr-line>Hunan</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Collaco</surname><given-names>Bernardo G</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Akbar</surname><given-names>Natasha</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Guoping Yang, Prof Dr, Center of Clinical Pharmacology, The Third Xiangya Hospital, Central South University, No 138 Tongzipo Road, Yuelu District,Changsha, Hunan, 410013, China, 86 0731 88618933; <email>ygp9880@126.com</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>10</day><month>8</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e84086</elocation-id><history><date date-type="received"><day>15</day><month>09</month><year>2025</year></date><date date-type="rev-recd"><day>06</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>08</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Junlong Ma, Xuehong Wu, Zeying Feng, Yun Kuang, Zhendong Ding, Min Li, Guoping Yang. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 10.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e84086"/><abstract><sec><title>Background</title><p>Adverse drug events (ADEs) pose significant public health challenges and economic burdens. While substantial ADE information is documented in unstructured clinical notes, its extraction remains difficult due to semantic complexity. Large language models (LLMs) offer promising text comprehension capabilities but are often hindered by domain-specific hallucinations.</p></sec><sec><title>Objective</title><p>This study aims to evaluate the effectiveness of retrieval-augmented generation (RAG) in improving the identification of ADEs using LLMs from Chinese clinical narratives and to establish a paradigm for this task.</p></sec><sec sec-type="methods"><title>Methods</title><p>We collected and preprocessed 19,983 Chinese clinical notes, retaining 18,432 high-quality records. Following a rigorous annotation and deduplication process, we established a gold-standard reference dataset (n=2510) and an ADE knowledge base (n=5144) using a standardized JSON schema. We evaluated 3 state-of-the-art LLMs (DeepSeek-V3 [DeepSeek], ERNIE 3.5-8K [Baidu], and GPT-4o [OpenAI]) under 3 prompt strategies: nonaugmented generation (NAG), static-augmented generation (SAG), and RAG. Performance was comprehensively assessed using precision, recall, and <italic>F</italic><sub>1</sub>-score across 3 recognition matching levels (L1 exact, L2 sentence, and L3 overlap) via 1000 bootstrap resamples. Model robustness was further validated from real-world clinical progress notes, reflecting real-world ADE prevalence.</p></sec><sec sec-type="results"><title>Results</title><p>We successfully constructed and publicly released the first Chinese ADE corpus derived from clinical notes. Across the tested LLMs, RAG yielded higher <italic>F</italic><sub>1</sub>-scores than NAG and SAG at the L3 level. The optimal configuration, DeepSeek-V3 with RAG, achieved an overall L3-level <italic>F</italic><sub>1</sub>-score of 0.9638 (95% CI 0.9541&#x2010;0.9727). Notably, the RAG approach increased the recall of GPT-4o from 0.6419 under NAG to 0.9241 under RAG (FDR <italic>P</italic>=.003). Evaluation on real-world datasets demonstrated clinical utility, with the RAG prompt maintaining high discriminatory capability (specificity: 0.9821; <italic>F</italic><sub>2</sub>-score: 0.8885). Error analysis revealed that RAG successfully resolved common identification errors, both omissions and commissions, that were intractable for nonaugmented models.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Synergizing a curated domain-specific knowledge base with LLMs via a RAG architecture is an effective strategy for accurately identifying ADEs in unstructured Chinese clinical notes. This approach can mitigate hallucinations in LLMs, providing a foundational open-source benchmark and a robust technical framework to advance pharmacovigilance, drug safety research, and clinical decision support.</p></sec></abstract><kwd-group><kwd>adverse drug event</kwd><kwd>large language model</kwd><kwd>retrieval-augmented generation</kwd><kwd>Chinese clinical notes</kwd><kwd>pharmacovigilance</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Adverse drug events (ADEs) are defined as harmful reactions that are unrelated to the intended therapeutic effects of medications when they are administered at standard dosages and according to standard regimens. ADEs constitute a significant global public health concern. They are among the main causes of hospitalization and mortality in both developed and developing countries [<xref ref-type="bibr" rid="ref1">1</xref>]. Meta-analyses reveal that approximately 5%&#x2010;10% of patients in health care institutions experience ADEs [<xref ref-type="bibr" rid="ref2">2</xref>]. Furthermore, the economic burden attributable to ADEs within health care systems worldwide exceeds US $42 billion annually [<xref ref-type="bibr" rid="ref3">3</xref>]. The precise identification and comprehensive evaluation of ADEs, including elucidating their underlying etiologies, are imperative for mitigating harm and augmenting the quality of clinical care [<xref ref-type="bibr" rid="ref4">4</xref>]. Consequently, the surveillance and management of ADEs have emerged as critical public health priorities, wherein the accurate extraction of ADE-related information can significantly enhance drug safety and foster rational pharmacotherapy.</p><p>A substantial proportion of ADE information remains embedded within unstructured, narrative clinical notes, presenting formidable challenges to conventional manual review and extraction methodologies, which are often inefficient and labor-intensive [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>]. The advent of large language model (LLM)&#x2013;based tools, such as ChatGPT, offers a promising way to efficiently and accessibly identify and retrieve ADE information [<xref ref-type="bibr" rid="ref7">7</xref>]. Trained on vast corpora of textual data, LLMs demonstrate remarkable abilities in cross-domain text comprehension, logical inference, and human-like natural language generation [<xref ref-type="bibr" rid="ref8">8</xref>]. Empirical studies have demonstrated their utility across diverse medical applications, including disease diagnosis and management [<xref ref-type="bibr" rid="ref9">9</xref>], patient education and counseling [<xref ref-type="bibr" rid="ref10">10</xref>], clinical text analysis [<xref ref-type="bibr" rid="ref11">11</xref>], and postoperative risk stratification [<xref ref-type="bibr" rid="ref12">12</xref>]. Despite growing evidence attesting to the value of LLMs in multiple medical domains, their potential in pharmacovigilance, particularly in the detection of ADEs in clinical notes, remains insufficiently explored. Furthermore, as LLMs are primarily pretrained on publicly available datasets without domain-specific clinical fine-tuning, they are prone to generating &#x201C;hallucinations,&#x201D; a phenomenon warranting heightened vigilance in health care contexts [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref14">14</xref>]. For instance, Williams et al [<xref ref-type="bibr" rid="ref15">15</xref>] found that GPT-4 and GPT-3.5-turbo models produced fabricated patient visit summaries at an alarming rate of up to 42%.</p><p>Retrieval-augmented generation (RAG) architectures enhance the performance of LLMs by integrating external information retrieval mechanisms, thus improving response accuracy and practical applicability [<xref ref-type="bibr" rid="ref16">16</xref>]. By dynamically combining domain-specific knowledge bases with user queries, RAG provides comprehensive, high-quality contextual information, reducing the likelihood of erroneous model outputs and effectively addressing the hallucination challenge [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>]. Additionally, RAG affords access to the latest reliable knowledge without the substantial costs of extensive model fine-tuning [<xref ref-type="bibr" rid="ref19">19</xref>]. Recent studies have highlighted RAG&#x2019;s superiority over standard LLMs in biomedical tasks, including question answering, text and image generation, and clinical scenario interpretation [<xref ref-type="bibr" rid="ref20">20</xref>]. For instance, Li et al [<xref ref-type="bibr" rid="ref21">21</xref>] showed that combining RAG with LLMs substantially improves the accuracy and reliability of COVID-19 fact-checking, successfully overcoming the inherent hallucination and context-inaccuracy issues [<xref ref-type="bibr" rid="ref21">21</xref>]. Nonetheless, the efficacy of RAG depends on the availability of trustworthy, unbiased data, and the quality of external domain knowledge significantly affects performance [<xref ref-type="bibr" rid="ref22">22</xref>]. Currently, ADE-related knowledge is fragmented and lacks structured representation, posing significant barriers to the deployment of RAG and LLM frameworks for ADE extraction. Therefore, there is a compelling need to construct a high-quality, structured Chinese ADE corpus.</p><p>This study aims to evaluate the effectiveness of RAG in enhancing LLM-based identification of ADE information within Chinese clinical notes. By integrating a curated knowledge base and using several foundational LLMs, this research systematically assesses multiple RAG-based architectures. Through comparative analysis of model performance under various configurations, this study seeks to establish an optimal paradigm for extracting ADE information from clinical notes, thereby providing robust technical support for pharmacovigilance and rational pharmacotherapy.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Datasets and Data Preprocessing</title><p>The study flow is illustrated in <xref ref-type="fig" rid="figure1">Figure 1</xref>. A total of 19,983 clinical notes were randomly collected from multiple large-scale tertiary hospitals, encompassing records from the years 2007 through 2023. These records covered a variety of document types, including ward round notes, initial clinical summaries, transfusion logs, handover reports, and routine progress documentation. A rigorous data preprocessing pipeline was implemented to protect patient confidentiality and improve data integrity. During the deidentification phase, all personally identifiable information relating to patients and physicians was uniformly replaced with generic terms, such as &#x201C;Patient&#x201D; and &#x201C;Physician.&#x201D; Similarly, health care institution names were anonymized.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Workflow of the adverse drug event recognition study. ADE: adverse drug event; LLM: large language model; NAG: nonaugmented generation; RAG: retrieval-augmented generation; SAG: static-augmented generation.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e84086_fig01.png"/></fig><p>To ensure data quality, a multitiered filtration protocol was implemented: (1) removal of records manifesting abnormal, nonsensical, or noncompliant content with established medical documentation standards; (2) exclusion of entries shorter than 100 characters; and (3) elimination of excessively long records exceeding 3000 characters that contained abundant irrelevant content or extraneous tags. Following this systematic preprocessing, 18,450 records meeting the quality criteria were initially retained. To prevent potential patient-level data leakage, we further verified all records using a double-deidentified patient master index. This verification showed that the 18,450 notes corresponded to 18,432 unique patients, with only 14 patients contributing multiple notes. After retaining 1 note per patient and removing 18 duplicate-patient records, all of which were ADE-negative, 18,432 patient-level clinical records were retained.</p></sec><sec id="s2-2"><title>Dataset Partitioning and Annotation Generation</title><p>The curated dataset was randomly partitioned into 2 distinct subsets with a 3:7 ratio: (1) a standard reference dataset comprising 5535 entries, designated as the gold standard to rigorously validate the efficacy of ADE identification methodologies; (2) an ADE case knowledge base consisting of 12,897 entries, intended for providing domain-specific knowledge. Both datasets were processed using a standardized annotation protocol. To eliminate potential automation bias, initial ADE preannotations were generated using Qwen-turbo (model version updated April 28, 2025; temperature=1.0, top_<italic>P</italic>=.9, penalty_score=1.05), a model explicitly distinct from the 3 evaluated LLMs. These preannotations were subsequently reviewed and revised by clinical pharmacists with 2&#x2010;10 years of pharmacovigilance experience. All annotators were fully blinded to the preannotation model and the evaluation design, receiving only raw clinical notes and machine-generated labels. Interrater reliability was assessed using Cohen &#x03BA; based on 5535 clinical notes independently reviewed by 2 annotators, yielding a high agreement (Cohen &#x03BA;=0.86). Any disagreements between annotators were resolved through consensus discussion, thereby confirming the high consistency and robustness of the gold-standard reference dataset.</p><p>During the annotation and review process, we observed a high degree of homogeneity in both clinical notes and the ADE annotations within the positive samples. Such redundancy has the potential to introduce bias in subsequent analyses, particularly as cases sharing identical ADE categories may collectively succeed or fail in recognition, thereby compromising the robustness of evaluative conclusions. To address this issue, we implemented a strict deduplication and balancing procedure to derive the final datasets. First, within the initial standard reference dataset of 5535 entries (comprising 2789 positive and 2746 negative instances), we deduplicated the positive reference set by limiting each drug entity to no more than 3 appearances. Deduplication was performed strictly according to the annotated drug entity name, without collapsing specific drugs into a generic category. Following deduplication, the positive reference set was refined to 1228 records. Correspondingly, an equivalent sample of 1282 instances was randomly selected from the 2746 negative cases, resulting in a balanced, final standard reference dataset of 2510 records. Subsequently, a similar deduplication and balancing process was applied to the initial dataset comprising 12,897 clinical records. This resulted in the creation of a final ADE knowledge base comprising 5144 records (2273 positive cases and 2871 negative cases).</p><p>To standardize the annotation process and accurately capture the subtle characteristics of ADE descriptions in clinical notes, a JSON schema was adopted for ADE annotation and recognition tasks (Figure S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Specifically, the JSON schema encapsulates a structured array wherein each ADE incident is described as an object comprising three pivotal fields: (1) sentence: the original textual fragment characterizing the ADE event; (2) drugs: an array listing the implicated pharmaceutical agents, accommodating polypharmacy interactions; and (3) reactions: an array detailing the corresponding ADE entities, supporting multiple concurrent reactions. If no ADE is identified within a record, an empty array &#x201C;[]&#x201D; is returned.</p><p>Using a JSON-based model for ADE information has several distinct advantages: (1) structured storage facilitates efficient data organization through standardized fields, thereby providing a solid foundation for downstream tasks such as causal inference and association analysis; (2) stringent field constraints (eg, requiring nonempty reaction arrays) enforce rigorous accuracy requirements for ADE entity recognition; and (3) compatibility with LLM input-output paradigms enhances the efficacy of generative models in extracting and completing ADE-related information.</p></sec><sec id="s2-3"><title>Model Setting and Prompt Design</title><p>The study used 3 state-of-the-art LLMs: DeepSeek-V3 (DeepSeek), ERNIE 3.5-8K (Baidu), and GPT-4o (OpenAI). Specifically, DeepSeek-V3 offers cutting-edge open-source performance with privacy-preserving local deployment; ERNIE 3.5-8K provides a commercial baseline optimized for Chinese clinical natural language processing; and GPT-4o serves as a top-tier, general-purpose international benchmark. Together, they represent a broad spectrum of open-source versus closed-source, Chinese-specialized versus general-purpose, and local versus cloud paradigms. The key to leveraging these LLMs was prompt engineering. To systematically assess the impact of prompt design and RAG strategy on ADE recognition performance, a progressive optimization framework was implemented: (1) nonaugmented generation (NAG)&#x2014;a zero-shot prompting baseline strategy supplying the LLM with ADE identification instructions without any supplementary knowledge, thereby delineating the model&#x2019;s innate capabilities; (2) static-augmented generation (SAG)&#x2014;building upon NAG, this zero-shot approach incorporates static clinical knowledge descriptions of ADE concepts to formulate semantically constrained recognition prompts, evaluating the influence of conceptual knowledge injection; and (3) RAG&#x2014;further advancing SAG into a dynamic few-shot prompting strategy with a dynamic retrieval mechanism that fetches analogous cases from the ADE knowledge base in real time, yielding context-aware prompts optimized for the complexity of clinical notes (<xref ref-type="fig" rid="figure2">Figure 2</xref> and Figure S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Prompt engineering framework for ADE recognition, translated into English for reader comprehension. The schematic details 3 step-wise augmentation strategies for large language models. (A) Nonaugmented generation serving as a zero-shot baseline. (B) Static-augmented generation integrating static clinical diagnostic rules. (C) Retrieval-augmented generation dynamically injecting top 5 similar cases for context-aware, few-shot prompting. ADE: adverse drug event; LLM: large language model; NAG: nonaugmented generation; RAG: retrieval-augmented generation; SAG: static-augmented generation.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e84086_fig02.png"/></fig><p>According to the retrieval schema of the RAG framework proposed in this paper, each case is structured as a key-value pair: the key is the original ADE text snippet, and the value is a standardized JSON string. Keys were vectorized using the shaw/dmeta-embedding-zh model (vector dimension=768) and indexed in a Milvus (v2.5.2; Zilliz) database, with corresponding JSONs stored as structured metadata. During dynamic retrieval, input clinical notes are split into sentences using punctuation marks and processed via a sliding window (size=3, step=2). These contextual segments are vectorized to retrieve the top 5 most similar ADE entries from the Milvus database via cosine similarity. The retrieval module achieved a Recall@5 of 0.908 and a mean reciprocal rank of 0.874, empirically verifying its ability to accurately fetch relevant ADE cases and reduce model hallucinations.</p><p>This prompt architecture follows a logical progression: from baseline capability verification to knowledge-enhanced refinement and adaptive contextualization, establishing a rigorous and reusable experimental paradigm for quantitatively evaluating knowledge augmentation strategies. All LLM inference, prompting, and ADE extraction tasks were conducted natively in Chinese. The English prompt text and translations shown in figures and tables are provided solely for reader comprehension and were not used as model inputs. The NAG instruction comprises 3 sections: &#x201C;System Instruction,&#x201D; &#x201C;Output Requirement,&#x201D; and &#x201C;Clinical Course Record.&#x201D; The &#x201C;System Instruction&#x201D; guides ADE identification, while the &#x201C;Output Requirement&#x201D; specifies adherence to the ADE JSON schema. The &#x201C;Clinical Course Record&#x201D; represents the data to be analyzed, with &#x201C;record&#x201D; acting as a placeholder for later use. The SAG instruction mirrors NAG but elaborates on ADE criteria, including temporal associations, drug exposure history, clinical manifestations, and causality terms. Exclusion criteria are also defined, including mentions of drugs without adverse reactions, non&#x2013;drug-related reactions, prophylactic medications, vigilant use, postdiscontinuation reactions, and prior ADEs. The RAG instruction extends SAG by adding a &#x201C;Reference Knowledge&#x201D; section. Specifically, the retrieved top 5 ADE reference cases are explicitly injected into this section of the structured RAG prompt template (<xref ref-type="fig" rid="figure2">Figure 2</xref>), providing contextual domain knowledge to enhance the LLMs&#x2019; accurate and interpretable ADE identification.</p></sec><sec id="s2-4"><title>Experimental Setup</title><p>Our entire experimental pipeline was implemented natively in Java (JDK 1.8; Oracle Corporation), using the HttpClient library to establish seamless communication with cloud-based model services. Although DeepSeek-V3 permits privacy-preserving local deployment, the present experiments used its cloud API version, together with the cloud APIs of ERNIE 3.5-8K and GPT-4o, to ensure consistent batch inference and experimental efficiency. All API calls used deidentified Chinese clinical-note inputs and Chinese prompt templates. To guarantee complete study reproducibility, the exact versions and configurations of the invoked models are specified as follows: DeepSeek-V3 (updated March 24, 2025; temperature=0.3, top_<italic>P</italic>=.70, penalty_score=1.0), ERNIE 3.5-8K (updated December 22, 2024; temperature=0.3, top_<italic>P</italic>=.70, penalty_score=1.0), and GPT-4o (updated November 20, 2024; temperature=0.3, top_<italic>P</italic>=.70, frequency_penalty=0, presence_penalty=0).</p></sec><sec id="s2-5"><title>Statistical Metrics for Model Evaluation</title><p>The experimental outcomes were evaluated using precision, recall, the <italic>F</italic><sub>1</sub>-score, specificity, the <italic>F</italic><sub>2</sub>-score, G-mean, and the Cohen &#x03BA; metric. Within the reference dataset, clinical notes containing ADEs were designated as positive instances, whereas those devoid of ADEs were classified as negative instances. The metrics were computed as follows:</p><disp-formula id="equWL1"><mml:math id="eqn1"><mml:mtext>Precision = </mml:mtext><mml:mfrac><mml:mrow><mml:mtext>TP </mml:mtext></mml:mrow><mml:mrow><mml:mtext>TP+FP</mml:mtext></mml:mrow></mml:mfrac></mml:math></disp-formula><disp-formula id="equWL2"><mml:math id="eqn2"><mml:mtext>Recall = </mml:mtext><mml:mfrac><mml:mrow><mml:mtext>TP</mml:mtext></mml:mrow><mml:mrow><mml:mtext>TP+FN</mml:mtext></mml:mrow></mml:mfrac></mml:math></disp-formula><disp-formula id="equWL3"><mml:math id="eqn3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mrow><mml:mi mathvariant="italic">F</mml:mi></mml:mrow><mml:mn>1</mml:mn></mml:msub><mml:mtext>-score</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mtext>2&#x00D7;Precision&#x00D7;Recall</mml:mtext><mml:mtext>Precision + Recall</mml:mtext></mml:mfrac></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><disp-formula id="equWL4"><mml:math id="eqn4"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mrow><mml:mi mathvariant="italic">F</mml:mi></mml:mrow><mml:mn>2</mml:mn></mml:msub><mml:mtext>-score</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>5</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mtext>Precision</mml:mtext><mml:mo>&#x00D7;</mml:mo><mml:mtext>Recall</mml:mtext></mml:mrow><mml:mrow><mml:mn>4</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mtext>Precision</mml:mtext><mml:mo>+</mml:mo><mml:mtext>Recall</mml:mtext></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><disp-formula id="equWL5"><mml:math id="eqn5"><mml:mi mathvariant="normal">S</mml:mi><mml:mi mathvariant="normal">p</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">f</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">y</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi mathvariant="normal">T</mml:mi><mml:mi mathvariant="normal">N</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">T</mml:mi><mml:mi mathvariant="normal">N</mml:mi><mml:mo>+</mml:mo><mml:mi mathvariant="normal">F</mml:mi><mml:mi mathvariant="normal">P</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula><disp-formula id="equWL6"><mml:math id="eqn6"><mml:mi mathvariant="normal">G</mml:mi><mml:mo>_</mml:mo><mml:mi mathvariant="normal">M</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mo>=</mml:mo><mml:msqrt><mml:mi mathvariant="normal">R</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">l</mml:mi><mml:mi mathvariant="normal">l</mml:mi><mml:mi mathvariant="normal">*</mml:mi><mml:mi mathvariant="normal">S</mml:mi><mml:mi mathvariant="normal">p</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">f</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">y</mml:mi><mml:mi mathvariant="normal"> </mml:mi></mml:msqrt></mml:math></disp-formula><disp-formula id="equWL7"><mml:math id="eqn7"><mml:msub><mml:mrow><mml:mi mathvariant="normal">P</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">o</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi mathvariant="normal">T</mml:mi><mml:mi mathvariant="normal">P</mml:mi><mml:mo>+</mml:mo><mml:mi mathvariant="normal">T</mml:mi><mml:mi mathvariant="normal">N</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">N</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula><disp-formula id="equWL8"><mml:math id="eqn8"><mml:msub><mml:mrow><mml:mi mathvariant="normal">P</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">e</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo>(</mml:mo><mml:mi mathvariant="normal">T</mml:mi><mml:mi mathvariant="normal">P</mml:mi><mml:mo>+</mml:mo><mml:mi mathvariant="normal">F</mml:mi><mml:mi mathvariant="normal">N</mml:mi><mml:mo>)</mml:mo><mml:mi mathvariant="normal">*</mml:mi><mml:mo>(</mml:mo><mml:mi mathvariant="normal">T</mml:mi><mml:mi mathvariant="normal">P</mml:mi><mml:mo>+</mml:mo><mml:mi mathvariant="normal">F</mml:mi><mml:mi mathvariant="normal">P</mml:mi><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mo>(</mml:mo><mml:mi mathvariant="normal">F</mml:mi><mml:mi mathvariant="normal">P</mml:mi><mml:mo>+</mml:mo><mml:mi mathvariant="normal">T</mml:mi><mml:mi mathvariant="normal">N</mml:mi><mml:mo>)</mml:mo><mml:mi mathvariant="normal">*</mml:mi><mml:mo>(</mml:mo><mml:mi mathvariant="normal">F</mml:mi><mml:mi mathvariant="normal">N</mml:mi><mml:mo>+</mml:mo><mml:mi mathvariant="normal">T</mml:mi><mml:mi mathvariant="normal">N</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi mathvariant="normal">N</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:math></disp-formula><disp-formula id="equWL9"><mml:math id="eqn9"><mml:mi mathvariant="normal">K</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="normal">P</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">o</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="normal">P</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">e</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="normal">P</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">e</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:math></disp-formula><p>TP denotes the count of true positives where the model correctly identifies an instance as positive; FP represents false positives where the model identifies an instance as positive but the reference set labels it as negative; FN signifies false negatives where the model fails to identify an instance marked positive in the reference set. P<sub>o</sub> represents the actual observed accuracy and P<sub>e</sub> represents the chance-expected accuracy. All reported metrics were computed at the document level using a clinical note, rather than an individual ADE entity, as the unit of evaluation.</p><p>It is important to recognize that identifying ADEs is a complex task. For any given sample in the standard reference set, a more nuanced determination of recognition accuracy is required. To this end, we have defined 3 levels of recognition precision: L1, L2, and L3 (<xref ref-type="table" rid="table1">Table 1</xref>). The L1, L2, and L3 rules were used to determine TP versus FN status only among ADE-positive notes. L1 (exact match) strictly requires complete alignment of the predicted sentence, drug array, and adverse reaction array with the gold standard. Under L1, partial matches (eg, correct drug but incomplete reactions) were rejected as TP and counted as FN at the document level. Conversely, L2 (sentence-level match) counted a note as TP when the predicted and gold-standard sentences aligned, even if the drug or reaction entities were incomplete or partially discrepant. L3 (overlap match) further relaxes this criterion, counting a TP if at least one drug or adverse reaction entity overlaps, regardless of overall array completeness. Figure S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> provides illustrative examples to further elucidate these definitions. Paired bootstrap tests were applied to the 1000 bootstrap resamples to evaluate the statistical significance of performance differences across models, prompts, and matching levels. <italic>P</italic> values were adjusted using the Benjamini-Hochberg false discovery rate (FDR) procedure, and statistical significance was defined as FDR <italic>P</italic>&#x003C;.05.</p><p>To evaluate model robustness under real-world clinical conditions, we additionally assessed DeepSeek-V3 using uncurated clinical progress notes sampled from the original pool of 19,983 raw records. Prior to calculating the metrics, the remaining records that were excluded from the curated dataset were also annotated with supplementary ADE labels using the same JSON schema and a consistent manual review protocol. In total, 1000 resampling iterations were performed, with 1000 notes randomly selected in each iteration and evaluated under the NAG, SAG, and RAG prompting strategies. The average ADE-positive prevalence was approximately 4.6%. Given this imbalance, specificity, <italic>F</italic><sub>2</sub>-score, G-mean, and Cohen &#x03BA; were calculated in addition to precision, recall, and <italic>F</italic><sub>1</sub>-score to assess false-positive control, recall-oriented detection, overall discrimination, and agreement with the gold standard. To further assess performance on challenging clinical language, we constructed a targeted 100-note subset containing ADE negation, abbreviations, temporal and causal contexts, and evaluated it using DeepSeek-V3 with RAG. All reported metrics, including precision, recall, <italic>F</italic><sub>1</sub>-score, specificity, <italic>F</italic><sub>2</sub>-score, G-mean, and Cohen &#x03BA;, were computed at the document level using a clinical note.</p><p>Statistical analyses were performed in Python (version 3.10; Python Software Foundation) using <italic>NumPy</italic>, <italic>pandas</italic>, <italic>SciPy</italic>, and <italic>scikit-learn</italic> packages for bootstrap resampling, percentile-based 95% CIs, paired bootstrap tests, Cohen &#x03BA;, and classification metrics.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Definition of identification precision level<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Precision level</td><td align="left" valign="bottom">Description</td></tr></thead><tbody><tr><td align="left" valign="top">L1</td><td align="left" valign="top">The identified ADEs<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> are completely consistent with the annotated ADEs, including the number of identified ADEs, as well as the sentences, drugs, and reactions within each ADE.</td></tr><tr><td align="left" valign="top">L2</td><td align="left" valign="top">The identified ADEs are roughly consistent with the annotated ADEs. The number of ADEs matches, and the sentences within the ADEs are the same, but there are discrepancies in the drug and reaction entities.</td></tr><tr><td align="left" valign="top">L3</td><td align="left" valign="top">The number of identified ADEs does not match the number of annotated ADEs, but there is an overlap between them.</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>L1 (exact match of sentences, drugs, and reactions), L2 (sentence-level match with drug or reaction discrepancies), and L3 (partial overlap of entities).</p></fn><fn id="table1fn2"><p><sup>b</sup>ADE: adverse drug event.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-6"><title>Ethical Considerations</title><p>This retrospective study was approved by the Ethics Review Board of Xiangya Hospital of Central South University (2025091288). The requirement for informed consent was waived because this study involved a secondary analysis of existing clinical notes. All data used were anonymized to ensure participant privacy. No personally identifiable information was included in the study, nor was any such information disclosed to any LLMs.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Construction Results of the Standard Reference Set and ADE Knowledge Base</title><p>After a thorough manual review and annotation process, an initial analysis was conducted prior to data deduplication to understand the distribution characteristics of the data. A frequency analysis of drug entities and ADEs was conducted within the positive reference set. This dataset encompassed 748 unique drug entities, each of which appeared on average 3 times, with 28 entities exceeding 10 occurrences. Notably, chemotherapeutic agents represented the most common drug entity category, appearing 775 times (<xref ref-type="fig" rid="figure3">Figure 3A</xref>). This count included only instances in which the generic term itself was recorded in the annotation, whereas specific chemotherapy drugs were retained as distinct entities rather than aggregated under this term. Concurrently, 2668 unique ADE types were annotated, 22 of which appeared 10 or more times (<xref ref-type="fig" rid="figure3">Figure 3B</xref>). As detailed in the &#x201C;Methods&#x201D; section, to mitigate potential bias caused by this high frequency of redundant cases, both datasets underwent a rigorous deduplication and balancing process. This procedure ultimately yielded a refined standard reference dataset of 2510 records and an ADE knowledge base of 5144 records, which were subsequently used for all downstream model evaluations. A baseline RAG experiment using the full nondeduplicated knowledge base of 12,897 records showed slightly lower performance than the deduplicated knowledge base (Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), supporting that limiting each drug entity to no more than 3 appearances improved retrieval diversity and reduced retrieval homogenization.</p><p>This work marks the first public release of a Chinese clinical note-based ADE research dataset, which is intended to be an open resource for the scientific community. The dataset is accessible at a GitHub repository [<xref ref-type="bibr" rid="ref23">23</xref>]. Each record is characterized by 3 attributes: ID (a unique dataset identifier), Content (the clinical progress note text), and ADEs (annotations conforming to the ADE JSON schema specification).</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Frequency distribution of (A) drug entities and (B) adverse drug events in the initial positive reference set.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e84086_fig03.png"/></fig></sec><sec id="s3-2"><title>Overall Results</title><p>To ensure the reliability of the experimental results and the stability of the methodology, we used 1000 bootstrap resamples. In each round, 1000 samples were randomly selected with replacement from the standard reference dataset of 2510 samples, and the ADE identification method was evaluated on the resampled subset. The statistical results are summarized in <xref ref-type="table" rid="table2">Table 2</xref> as bootstrap means with percentile-based 95% CI. Under the NAG prompt, all LLMs demonstrated strong ADE recognition performance, with both DeepSeek-V3 and ERNIE 3.5-8K achieving an overall <italic>F</italic><sub>1</sub>-score of over 0.9. At the L2 and L3 matching levels, adoption of the SAG prompt improved the recognition performance of all models compared to the NAG prompt, indicating that optimizing prompt design and incorporating domain-specific knowledge generally enhances the model&#x2019;s ability to recognize ADEs. Notably, applying the RAG prompt, which integrates retrieval information from a large-scale ADE case knowledge base, resulted in even greater performance improvements for all models compared to the SAG prompt. DeepSeek-V3 achieved the most outstanding recognition performance with an impressive overall <italic>F</italic><sub>1</sub>-score of 0.9638 (95% CI 0.9541&#x2010;0.9727).</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Overall performance comparison of various LLMs<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> on different matching levels and prompting strategies<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Prompt, LLM, and level</td><td align="left" valign="bottom">Precision, mean (95% CI)</td><td align="left" valign="bottom">Recall, mean (95% CI)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score, mean (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">NAG<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ERNIE 3.5-8K</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L1</td><td align="left" valign="top">0.9494 (0.9317&#x2010;0.9678)</td><td align="left" valign="top">0.5951 (0.5620&#x2010;0.6260)</td><td align="left" valign="top">0.7315 (0.7054&#x2010;0.7554)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L2</td><td align="left" valign="top">0.9659 (0.9539&#x2010;0.9783)</td><td align="left" valign="top">0.8968 (0.8780&#x2010;0.9160)</td><td align="left" valign="top">0.9300 (0.9174&#x2010;0.9420)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L3</td><td align="left" valign="top">0.9669 (0.9553&#x2010;0.9789)</td><td align="left" valign="top">0.9247 (0.9080&#x2010;0.9420)</td><td align="left" valign="top">0.9453 (0.9345&#x2010;0.9560)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek-V3</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L1</td><td align="left" valign="top">0.9288 (0.9061&#x2010;0.9534)</td><td align="left" valign="top">0.5143 (0.4800&#x2010;0.5480)</td><td align="left" valign="top">0.6619 (0.6305&#x2010;0.6928)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L2</td><td align="left" valign="top">0.9509 (0.9358&#x2010;0.9674)</td><td align="left" valign="top">0.7642 (0.7340&#x2010;0.7920)</td><td align="left" valign="top">0.8473 (0.8271&#x2010;0.8666)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L3</td><td align="left" valign="top">0.9559 (0.9425&#x2010;0.9707)</td><td align="left" valign="top">0.8544 (0.8300&#x2010;0.8780)</td><td align="left" valign="top">0.9022 (0.8870&#x2010;0.9179)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-4o</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L1</td><td align="left" valign="top">0.9395 (0.9158&#x2010;0.9637)</td><td align="left" valign="top">0.3947 (0.3620&#x2010;0.4280)</td><td align="left" valign="top">0.5556 (0.5230&#x2010;0.5895)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L2</td><td align="left" valign="top">0.9575 (0.9402&#x2010;0.9738)</td><td align="left" valign="top">0.5723 (0.5400&#x2010;0.6060)</td><td align="left" valign="top">0.7162 (0.6888&#x2010;0.7432)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L3</td><td align="left" valign="top">0.9619 (0.9470&#x2010;0.9767)</td><td align="left" valign="top">0.6419 (0.6100&#x2010;0.6720)</td><td align="left" valign="top">0.7698 (0.7454&#x2010;0.7934)</td></tr><tr><td align="left" valign="top">SAG<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ERNIE 3.5-8K</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L1</td><td align="left" valign="top">0.9178 (0.8944&#x2010;0.9403)</td><td align="left" valign="top">0.5674 (0.5340&#x2010;0.6020)</td><td align="left" valign="top">0.7011 (0.6733&#x2010;0.7282)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L2</td><td align="left" valign="top">0.9485 (0.9337&#x2010;0.9632)</td><td align="left" valign="top">0.9360 (0.9200&#x2010;0.9520)</td><td align="left" valign="top">0.9422 (0.9308&#x2010;0.9534)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L3</td><td align="left" valign="top">0.9499 (0.9358&#x2010;0.9641)</td><td align="left" valign="top">0.9644 (0.9520&#x2010;0.9780)</td><td align="left" valign="top">0.9571 (0.9475&#x2010;0.9663)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek-V3</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L1</td><td align="left" valign="top">0.9724 (0.9576&#x2010;0.9863)</td><td align="left" valign="top">0.5425 (0.5100&#x2010;0.5760)</td><td align="left" valign="top">0.6963 (0.6684&#x2010;0.7234)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L2</td><td align="left" valign="top">0.9821 (0.9722&#x2010;0.9908)</td><td align="left" valign="top">0.8471 (0.8240&#x2010;0.8720)</td><td align="left" valign="top">0.9096 (0.8942&#x2010;0.9237)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L3</td><td align="left" valign="top">0.9833 (0.9741&#x2010;0.9914)</td><td align="left" valign="top">0.9080 (0.8900&#x2010;0.9260)</td><td align="left" valign="top">0.9441 (0.9333&#x2010;0.9556)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-4o</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L1</td><td align="left" valign="top">0.9742 (0.9583&#x2010;0.9908)</td><td align="left" valign="top">0.4432 (0.4120&#x2010;0.4780)</td><td align="left" valign="top">0.6090 (0.5778&#x2010;0.6424)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L2</td><td align="left" valign="top">0.9821 (0.9709&#x2010;0.9937)</td><td align="left" valign="top">0.6440 (0.6120&#x2010;0.6760)</td><td align="left" valign="top">0.7778 (0.7534&#x2010;0.8019)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L3</td><td align="left" valign="top">0.9835 (0.9732&#x2010;0.9943)</td><td align="left" valign="top">0.6993 (0.6680&#x2010;0.7300)</td><td align="left" valign="top">0.8173 (0.7952&#x2010;0.8391)</td></tr><tr><td align="left" valign="top">RAG<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ERNIE 3.5-8K</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L1</td><td align="left" valign="top">0.9439 (0.9251&#x2010;0.9623)</td><td align="left" valign="top">0.5878 (0.5540&#x2010;0.6220)</td><td align="left" valign="top">0.7243 (0.6986&#x2010;0.7500)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L2</td><td align="left" valign="top">0.9639 (0.9520&#x2010;0.9765)</td><td align="left" valign="top">0.9316 (0.9140&#x2010;0.9500)</td><td align="left" valign="top">0.9474 (0.9370&#x2010;0.9586)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L3</td><td align="left" valign="top">0.9648 (0.9530&#x2010;0.9773)</td><td align="left" valign="top">0.9569 (0.9420&#x2010;0.9700)</td><td align="left" valign="top">0.9608 (0.9519&#x2010;0.9700)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek-V3</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L1</td><td align="left" valign="top">0.9686 (0.9525&#x2010;0.9851)</td><td align="left" valign="top">0.5482 (0.5160&#x2010;0.5820)</td><td align="left" valign="top">0.7000 (0.6727&#x2010;0.7270)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L2</td><td align="left" valign="top">0.9796 (0.9688&#x2010;0.9905)</td><td align="left" valign="top">0.8537 (0.8300&#x2010;0.8780)</td><td align="left" valign="top">0.9123 (0.8987&#x2010;0.9268)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L3</td><td align="left" valign="top">0.9816 (0.9716&#x2010;0.9916)</td><td align="left" valign="top">0.9466 (0.9320&#x2010;0.9620)</td><td align="left" valign="top">0.9638 (0.9541&#x2010;0.9727)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT-4o</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L1</td><td align="left" valign="top">0.9577 (0.9396&#x2010;0.9741)</td><td align="left" valign="top">0.5797 (0.5480&#x2010;0.6120)</td><td align="left" valign="top">0.7221 (0.6955&#x2010;0.7473)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L2</td><td align="left" valign="top">0.9710 (0.9586&#x2010;0.9819)</td><td align="left" valign="top">0.8565 (0.8340&#x2010;0.8800)</td><td align="left" valign="top">0.9101 (0.8958&#x2010;0.9241)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L3</td><td align="left" valign="top">0.9730 (0.9613&#x2010;0.9832)</td><td align="left" valign="top">0.9241 (0.9080&#x2010;0.9420)</td><td align="left" valign="top">0.9479 (0.9372&#x2010;0.9588)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup> LLM: large language model.</p></fn><fn id="table2fn2"><p><sup>b</sup>Values are presented as means with percentile-based 95% CIs from 1000 bootstrap resampling iterations. Prompting strategies evaluated include nonaugmented generation serving as a zero-shot baseline, static-augmented generation integrating static clinical diagnostic rules, and retrieval-augmented generation dynamically injecting the top 5 similar cases for context-aware, few-shot prompting. Matching levels are defined as L1 (exact match of sentences, drugs, and reactions), L2 (sentence-level match with drug or reaction discrepancies), and L3 (partial overlap of entities). </p></fn><fn id="table2fn3"><p><sup>c</sup> NAG: nonaugmented generation.</p></fn><fn id="table2fn4"><p><sup>d</sup>SAG: static-augmented generation.</p></fn><fn id="table2fn5"><p><sup>e</sup>RAG: retrieval-augmented generation.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Comparative Analysis of Prompts</title><p>Initially, an analysis of the ADE recognition instruction framework at the L3 level is presented. As illustrated in <xref ref-type="fig" rid="figure4">Figure 4</xref>, the comparison of recognition performance across various prompts at the L3 level is demonstrated for different LLMs. Compared with NAG, the SAG prompt increased the <italic>F</italic><sub>1</sub>-score of DeepSeek-V3 from 0.9022 to 0.9441 (FDR <italic>P</italic>=.003) and GPT-4o from 0.7698 to 0.8173 (FDR <italic>P</italic>=.006). For ERNIE 3.5-8K, SAG produced only a modest increase from 0.9453 to 0.9571, which did not reach statistical significance in the paired bootstrap test (FDR <italic>P</italic>=.09). Building on the SAG prompt, the RAG prompt further augmented the comprehension capacity of the LLMs, yielding remarkable performance. DeepSeek-V3 and GPT-4o showed improvements of 1.97% (FDR <italic>P</italic>=.003) and 13.06% (FDR <italic>P</italic>=.003), respectively. These findings affirm the crucial role of augmenting domain knowledge in improving ADE recognition performance within mainstream LLMs and validate the effectiveness of the ADE case knowledge base designed in this study. The knowledge retrieval enhancement method was successful in significantly boosting ADE recognition performance. Ultimately, under the RAG prompt, mainstream LLMs exhibited remarkable improvement in ADE recognition, with average <italic>F</italic><sub>1</sub>-scores exceeding 0.95.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Performance of large language models in adverse drug event recognition across prompting strategies. Bar charts illustrate precision, recall, and <italic>F</italic><sub>1</sub>-scores under nonaugmented generation, static-augmented generation, and retrieval-augmented generation prompts evaluated at the L3 matching level. Bars represent bootstrap mean values, and error bars indicate percentile-based 95% CIs. Knowledge augmentation significantly improves performance, with retrieval-augmented generation achieving average <italic>F</italic><sub>1</sub>-scores &#x003E;0.95 and substantially boosting GPT-4o&#x2019;s recall. NAG: nonaugmented generation; RAG: retrieval-augmented generation; SAG: static-augmented generation.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e84086_fig04.png"/></fig><p>Regarding precision (<xref ref-type="fig" rid="figure4">Figure 4</xref>), all models achieved a high level of precision under the NAG prompt, with the lowest precision reaching 95.59%. Minor fluctuations were subsequently observed under the SAG and RAG prompts, yet an overall upward trend in performance was evident. In terms of recall (<xref ref-type="fig" rid="figure4">Figure 4</xref>), ERNIE 3.5-8K and DeepSeek-V3 similarly achieved high recall rates under the NAG prompt, reaching 92.47% and 85.44%, respectively. In contrast, GPT-4o exhibited a significantly lower recall rate of only 64.19%, indicating a substantial proportion of ADEs went unrecognized. However, under the SAG prompt, GPT-4o&#x2019;s recall rate increased to 69.93%, and under the RAG prompt, it surged to 92.41%, representing an improvement of almost 30% (FDR <italic>P</italic>=.003). This corroborates the effectiveness of the knowledge retrieval enhancement method used in this experiment. It also highlights that while GPT-4o has considerable generative capabilities, its lack of domain-specific understanding limits its performance. However, upon incorporating contextual knowledge, its performance was substantially enhanced, ultimately bringing its overall <italic>F</italic><sub>1</sub>-score in line with those of ERNIE 3.5-8K and DeepSeek-V3.</p></sec><sec id="s3-4"><title>Comparative Analysis of LLMs</title><p>We further analyzed the variations in ADE recognition performance across different models. Taking the RAG prompt as an example, the comparative analysis at different matching levels is illustrated in <xref ref-type="fig" rid="figure5">Figure 5</xref>. With regard to precision, all 3 models demonstrated consistently stable performance across the L1, L2, and L3 levels, with each achieving a precision rate of over 94%. In terms of recall, ERNIE 3.5-8K slightly outperformed the other 2 models. Regarding the <italic>F</italic><sub>1</sub>-score, there was minimal variation in the models&#x2019; recognition performance: at the L1 level, ERNIE 3.5-8K marginally surpassed the others with an <italic>F</italic><sub>1</sub>-score of 72.43%, though this advantage was not statistically significant compared to GPT-4o (FDR <italic>P</italic>=.86); at the L2 level, ERNIE 3.5-8K led with a score of 94.74%, significantly outperforming both DeepSeek-V3 and GPT-4o (both FDR <italic>P</italic>=.003). At the L3 level, DeepSeek-V3 slightly outperformed the others with an <italic>F</italic><sub>1</sub>-score of 96.38% (FDR <italic>P</italic>=.01 compared to GPT-4o, but no significant difference compared to ERNIE 3.5-8K, FDR <italic>P</italic>=.68). Thus, under the RAG prompt, the overall performance of LLMs in ADE recognition for Chinese clinical progress notes is relatively consistent, with each model displaying robust semantic comprehension and generation capabilities.</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Performance comparison of various large language models. Bar charts compare precision, recall, and <italic>F</italic><sub>1</sub>-scores for 3 large language models using the retrieval-augmented generation prompt at L1, L2, and L3 levels. Bars represent bootstrap mean values, and error bars indicate percentile-based 95% CIs. For <italic>F</italic><sub>1</sub>-scores, ERNIE 3.5-8K marginally led at L1 (72.43%) and L2 (94.74%), whereas DeepSeek-V3 led at L3 (96.38%). LLM: large language model.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e84086_fig05.png"/></fig></sec><sec id="s3-5"><title>Comparative Analysis of Various Matching Levels</title><p>Taking the RAG prompt analysis as an example, we examined the models&#x2019; performance disparities across different matching levels (<xref ref-type="fig" rid="figure6">Figure 6</xref>). The performance of each model showed an upward trend from L1 to L2, and subsequently to L3 (all FDR <italic>P</italic>&#x003C;.05), with the most significant improvements in recall value. For instance, at L2, ERNIE 3.5-8K demonstrated a 34.38% improvement over L1, reaching 93.16%. DeepSeek-V3 showed a 30.55% increase, achieving 85.37%; and GPT-4o exhibited a 27.68% enhancement, attaining 85.65%. These results emphasize the important role of accurate entity recognition in ADE detection. Regarding the <italic>F</italic><sub>1</sub>-score, the substantial increase in recall at L2 led to a similar upward trend across all matching levels.</p><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Performance comparison of large language models on different matching levels. Performance of large language models across L1, L2, and L3 matching levels. Evaluated using the retrieval-augmented generation prompt, bar charts demonstrate a significant upward trend in precision, recall, and <italic>F</italic><sub>1</sub>-scores from L1 to L3. Bars represent bootstrap mean values, and error bars indicate percentile-based 95% CIs. Substantial recall improvements from L1 to L2 notably drive overall <italic>F</italic><sub>1</sub>-score increases.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e84086_fig06.png"/></fig><p>Given that the most substantial performance variance occurred at the L2 level, we undertook a granular analysis of these cases. Focusing on the DeepSeek-V3 model as a representative example, we identified 376 instances of L2-level matches within a standard reference corpus of 2510 entries. Our analysis of these instances revealed 2 main error categories: anomalous recognition of drug entities and anomalous recognition of adverse reaction entities (<xref ref-type="table" rid="table3">Table 3</xref>). The former category was further stratified into four subtypes: (1) omission of the correct drug entity, (2) commission of an incorrect drug entity, (3) misidentification of the drug entity, and (4) lack of precision in boundary detection. Similarly, anomalies in adverse reaction entity recognition comprised: (1) entity omission, (2) entity commission, (3) failure to disaggregate compound entities, and (4) imprecise boundary detection. Quantitative analysis of these error subtypes highlighted the omission of adverse reaction entities as the predominant failure mode, accounting for around 40.7% of the L2-level discrepancies.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Case analysis of anomalies in entity recognition.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Categories and specific reasons</td><td align="left" valign="bottom">Examples of cases</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Abnormal drug entities</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Missed entities</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>#13806<bold>:</bold> Failure to identify ARB<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup> drug entities</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Incorrect additional entities</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>#15018<bold>:</bold> Among the 2 identified drug entities, &#x201C;amino acids&#x201D; and &#x201C;compound amino acids,&#x201D; only &#x201C;compound amino acids&#x201D; is required.</p></list-item><list-item><p>#12731<bold>:</bold> An incorrect drug entity &#x201C;statin drugs&#x201D; has been identified.</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Misidentified</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>#7435<bold>:</bold> Two incorrect drug entities have been identified: &#x201C;reduced glutathione&#x201D; and &#x201C;magnesium isoglycyrrhizinate.&#x201D;</p></list-item><list-item><p>#13961<bold>:</bold> Two incorrect drug entities have been identified: &#x201C;methylprednisolone powder for injection&#x201D; and &#x201C;cyclosporine.&#x201D;</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Inaccurate entities</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>#4316<bold>:</bold> The identified drug entity is &#x201C;chemotherapy drugs,&#x201D; but it needs to be specified that the chemotherapy drug is &#x201C;ifosfamide.&#x201D;</p></list-item><list-item><p>#4176<bold>:</bold> The identified drug entity is &#x201C;hemostatic drugs,&#x201D; but it needs to be specified that the actual drug entity is &#x201C;pituitrin.&#x201D;</p></list-item><list-item><p>#7293<bold>:</bold> The identified drug entity is &#x201C;chemotherapy drugs,&#x201D; but it needs to be specified that the actual drug entities are &#x201C;oxaliplatin&#x201D; and &#x201C;fluorouracil.&#x201D;</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Abnormal reaction entities</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Missed entities</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>#12191<bold>:</bold> The adverse reaction entity &#x201C;elevated liver transaminases&#x201D; was not identified.</p></list-item><list-item><p>#6967<bold>:</bold> The adverse reaction entity &#x201C;elevated alanine aminotransferase&#x201D; was not identified.</p></list-item><list-item><p>#4782<bold>:</bold> The adverse reaction entity &#x201C;low platelet count&#x201D; was not identified.</p></list-item><list-item><p>#13993<bold>:</bold> The adverse reaction entity &#x201C;myalgia&#x201D; was not identified.</p></list-item><list-item><p>#14145<bold>:</bold> The adverse reaction entity &#x201C;rash&#x201D; was not identified.</p></list-item><list-item><p>#3341<bold>:</bold> The adverse reaction entity &#x201C;iodine allergy&#x201D; was not identified.</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Incorrect additional entities</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>#11640<bold>:</bold> An unconfirmed adverse reaction entity such as &#x201C;vomiting&#x201D; has been identified.</p></list-item><list-item><p>#14307<bold>:</bold> An unconfirmed adverse reaction entity, &#x201C;liver dysfunction,&#x201D; has been identified.</p></list-item><list-item><p>#3099<bold>:</bold> An unknown adverse reaction entity, &#x201C;allergic reaction,&#x201D; has been identified.</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Unsplited entities</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>#4444<bold>:</bold> The entity &#x201C;palpitations and shortness of breath&#x201D; has been identified and needs to be split into 2 separate entities: &#x201C;palpitations&#x201D; and &#x201C;shortness of breath.&#x201D;</p></list-item><list-item><p>#14307<bold>:</bold> The entity &#x201C;nausea and vomiting&#x201D; has been identified and requires splitting into 2 separate entities: &#x201C;nausea&#x201D; and &#x201C;vomiting.&#x201D;</p></list-item><list-item><p>#4698<bold>:</bold> The entity &#x201C;chills and fever&#x201D; has been identified and needs to be divided into two separate entities: &#x201C;chills&#x201D; and &#x201C;fever.&#x201D;</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Inaccurate entities</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>#4036<bold>:</bold> The entity &#x201C;abnormal liver function&#x201D; has been identified, but it needs to be clarified that the specific manifestations of this abnormal liver function are &#x201C;elevated alanine aminotransferase&#x201D; and &#x201C;elevated aspartate aminotransferase.&#x201D;</p></list-item><list-item><p>#3073<bold>:</bold> The adverse reaction entity description &#x201C;glucose+4&#x2265;55 mmol/L&#x201D; has been identified, but it needs to be converted into a clearly defined entity: &#x201C;elevated urinary glucose.&#x201D;</p></list-item></list></td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>ARB: angiotensin II receptor blocker.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-6"><title>Error Cases Analysis and Summary</title><p>A statistical analysis of error instances revealed that ERNIE 3.5-8K incorrectly handled 66 case records alone, DeepSeek-V3 incorrectly handled 38 case records alone, and GPT-4o incorrectly handled 73 case records alone (<xref ref-type="fig" rid="figure7">Figure 7</xref>). Notably, 15 critical records posed recognition challenges for all 3 models. This suggests that, despite the integration of 3 LLMs in a composite approach, these ADE remain unrecognized. A detailed etiological analysis of these 15 universally misidentified cases partitioned the errors into 2 principal categories: failures of omission (failure to identify a true ADE) and failures of commission (identification of an erroneous ADE; <xref ref-type="table" rid="table4">Table 4</xref>). The latter constituted the majority of errors, accounting for approximately 80% of instances. The primary drivers of these commission errors were identified as (1) non&#x2013;drug-related adverse events; (2) the incorrect classification of prophylactic, precautionary, or advisory medication mentions; and (3) the misinterpretation of historical ADE descriptions as contemporary events. Further investigation demonstrated that these erroneous identifications could be rectified by augmenting the models with a domain-specific ADE knowledge base via a RAG framework. This underscores the profound efficacy of incorporating a curated knowledge repository and RAG methods to enhance the precision of ADE detection.</p><fig position="float" id="figure7"><label>Figure 7.</label><caption><p>Venn diagram of overlapping recognition errors across 3 large language models.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e84086_fig07.png"/></fig><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Analysis of misidentifications by 3 LLMs<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup>.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Number</td><td align="left" valign="bottom">Case ID</td><td align="left" valign="bottom">Reasons for the errors</td></tr></thead><tbody><tr><td align="left" valign="top">1</td><td align="left" valign="top">#3160</td><td align="left" valign="top">Unrecognized ADE<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup>: &#x201C;Metabolic alkalosis with compensatory response, suspected diuretic-related.&#x201D;</td></tr><tr><td align="left" valign="top">2</td><td align="left" valign="top">#12251</td><td align="left" valign="top">Unrecognized ADE: &#x201C;The suspected cause of altered mental status includes intracranial edema or drug-induced toxicity combined with uremic toxin damage.&#x201D;</td></tr><tr><td align="left" valign="top">3</td><td align="left" valign="top">#3310</td><td align="left" valign="top">Unrecognized ADE: &#x201C;Cardiac troponin I negative, electrolytes normal; drug-related etiology suspected.&#x201D;</td></tr><tr><td align="left" valign="top">4</td><td align="left" valign="top">#4650</td><td align="left" valign="top">Misidentified ADE: &#x201C;The instructor noted that sodium nitroprusside had been administered for 10 days and discontinued to prevent long-term adverse effects, with isosorbide dinitrate substituted for vasodilation, antihypertension, and heart failure control.&#x201D; It pertains to prophylactic drug discontinuation due to anticipated adverse reactions, rather than an actual ADE.</td></tr><tr><td align="left" valign="top">5</td><td align="left" valign="top">#3896</td><td align="left" valign="top">Misidentified ADE: &#x201C;The patient received 300 mL of O(+) frozen plasma yesterday and reported generalized pruritus, pain, and numbness post-transfusion.&#x201D; It pertains to a transfusion-related reaction, not a drug-induced adverse event.</td></tr><tr><td align="left" valign="top">6</td><td align="left" valign="top">#11788</td><td align="left" valign="top">Misidentified ADE: &#x201C;E4A+CO2P: K+ 5.74 mmol/L indicates hyperkalemia, suspected to result from recent repeated blood transfusions.&#x201D; It pertains to a transfusion-associated electrolyte disturbance, not a drug-induced adverse event.</td></tr><tr><td align="left" valign="top">7</td><td align="left" valign="top">#4651</td><td align="left" valign="top">Misidentified ADE: &#x201C;Yesterday, during the infusion of 600 mL of O-Rh(D)-positive frozen plasma of the same blood type, the patient developed pruritus.&#x201D; It pertains to a transfusion-associated allergic reaction, not a drug-induced adverse event.</td></tr><tr><td align="left" valign="top">8</td><td align="left" valign="top">#12270</td><td align="left" valign="top">Misidentified ADE: &#x201C;Liver function tests revealed elevated transaminases; Shu Ganning Injection was administered for hepatic protection.&#x201D; It pertains to medication management rather than an ADE.</td></tr><tr><td align="left" valign="top">9</td><td align="left" valign="top">#4527</td><td align="left" valign="top">Misidentified ADE: &#x201C;Considering the neurotoxic potential of vincristine causing peripheral neuropathy (numbness in hands and feet), vitamin B1 tablets 20mg orally were added prophylactically today.&#x201D; It constitutes a description of prophylactic medication for adverse reaction prevention, rather than an actual ADE.</td></tr><tr><td align="left" valign="top">10</td><td align="left" valign="top">#11746</td><td align="left" valign="top">Misidentified ADE: &#x201C;Given the patient&#x2019;s clinical deterioration with potential pulmonary hemorrhage relapse,(LMWH)<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup> was discontinued, and airway status monitoring was initiated.&#x201D; It constitutes proactive ADE prevention measures rather than an actual ADE.</td></tr><tr><td align="left" valign="top">11</td><td align="left" valign="top">#7332</td><td align="left" valign="top">Misidentified ADE: &#x201C;Patient&#x2019;s family was informed of significant antifungal drug-related risks, including hepatorenal toxicity, leukopenia or thrombocytopenia, and adverse reactions such as chills, high fever, and thrombophlebitis.&#x201D; It constitutes pharmacovigilance counseling rather than an actual ADE.</td></tr><tr><td align="left" valign="top">12</td><td align="left" valign="top">#2939</td><td align="left" valign="top">Misidentified ADE: &#x201C;Patient demonstrated clinical improvement post-intravenous doxofylline and ambroxol administration (paroxysmal wheezing alleviated), with residual symptoms of mild white sputum production and pruritic trunk rash.&#x201D; It pertains to therapeutic response assessment rather than an ADE.</td></tr><tr><td align="left" valign="top">13</td><td align="left" valign="top">#5973</td><td align="left" valign="top">Misidentified ADE: &#x201C;Amikacin administered for 7 days; discontinued today to prevent potential renal impairment.&#x201D; It constitutes a prophylactic toxicity narrative rather than an ADE.</td></tr><tr><td align="left" valign="top">14</td><td align="left" valign="top">#11731</td><td align="left" valign="top">Misidentified ADE: &#x201C;Patient exhibits hypoglycemia during early morning and nighttime; insulin dose reduced to 10 units subcutaneous injection at bedtime today.&#x201D; The full clinical context did not provide sufficient causal evidence linking the hypoglycemia episode to insulin as a dose-dependent ADE.</td></tr><tr><td align="left" valign="top">15</td><td align="left" valign="top">#3074</td><td align="left" valign="top">Misidentified ADE: &#x201C;Patient on self-administered (ART)<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup> with history of (DILI)<sup><xref ref-type="table-fn" rid="table4fn5">e</xref></sup>.&#x201D; It represents a preexisting adverse drug reaction.</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup> LLM: large language model.</p></fn><fn id="table4fn2"><p><sup>b</sup>ADE: adverse drug event.</p></fn><fn id="table4fn3"><p><sup>c</sup>LMWH:  low-molecular-weight heparin.</p></fn><fn id="table4fn4"><p><sup>d</sup>ART: antiretroviral therapy.</p></fn><fn id="table4fn5"><p><sup>e</sup>DILI: drug-induced liver injury.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-7"><title>Subanalysis of Complex ADE Linguistic Contexts</title><p>To empirically evaluate the model&#x2019;s ability to handle complex clinical language, we constructed a targeted subset of 100 clinical notes containing 3 challenging ADE-related contexts: negation, abbreviations, and temporal and causal descriptions (Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Using DeepSeek-V3 under the RAG prompting strategy, the model achieved a precision of 0.9167, recall of 0.9429, and <italic>F</italic><sub>1</sub>-score of 0.9296 on this subset. These findings indicate that the RAG framework can effectively process negated ADE descriptions, abbreviated drug expressions, and historical or risk-monitoring contexts, supporting its robustness in complex clinical narratives.</p></sec><sec id="s3-8"><title>Performance Evaluation on Real-World Dataset</title><p>To validate clinical utility under real-world conditions, DeepSeek-V3 was evaluated using 1000 random resampling iterations from 19,983 uncurated clinical progress notes, with 1000 notes sampled in each iteration (<xref ref-type="table" rid="table5">Table 5</xref>). At the L3 level, RAG achieved an <italic>F</italic><sub>1</sub>-score of 0.8159 (95% CI 0.7736&#x2010;0.8598), compared with 0.6380 (95% CI 0.5759&#x2010;0.6942) for NAG and 0.8134 (95% CI 0.7600&#x2010;0.8627) for SAG. Despite relatively low precision under this low-prevalence setting, RAG demonstrated strong clinical utility, achieving a recall of 0.9448, <italic>F</italic><sub>2</sub>-score of 0.8885, G-mean of 0.9631, Cohen &#x03BA; of 0.8057, and specificity of 0.9821. These findings confirm that while absolute <italic>F</italic><sub>1</sub>-scores slightly adjust due to the natural data imbalance, integrating a curated knowledge base via RAG remains a highly effective and practically viable approach for clinical ADE identification.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Performance of DeepSeek-V3 on real-world uncurated clinical progress notes<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup>.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Prompt and level</td><td align="left" valign="bottom">Precision, mean (95% CI)</td><td align="left" valign="bottom">Recall, mean (95% CI)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score, mean (95% CI)</td><td align="left" valign="bottom">Specificity, mean (95% CI)</td><td align="left" valign="bottom"><italic>F</italic><sub>2</sub>-score, mean (95% CI)</td><td align="left" valign="bottom">G-mean, mean (95% CI)</td><td align="left" valign="bottom">Cohen &#x03BA;, mean (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="8">NAG<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L1</td><td align="left" valign="top">0.3836 (0.3051&#x2010;0.4603)</td><td align="left" valign="top">0.5151 (0.3696&#x2010;0.6522)</td><td align="left" valign="top">0.4391 (0.3400&#x2010;0.5345)</td><td align="left" valign="top">0.9602 (0.9539&#x2010;0.9665)</td><td align="left" valign="top">0.4816 (0.3571&#x2010;0.6009)</td><td align="left" valign="top">0.7015 (0.5960&#x2010;0.7931)</td><td align="left" valign="top">0.4080 (0.3055&#x2010;0.5079)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L2</td><td align="left" valign="top">0.4810 (0.4225&#x2010;0.5375)</td><td align="left" valign="top">0.7640 (0.6522&#x2010;0.8696)</td><td align="left" valign="top">0.5899 (0.5167&#x2010;0.6557)</td><td align="left" valign="top">0.9602 (0.9539&#x2010;0.9665)</td><td align="left" valign="top">0.6831 (0.5882&#x2010;0.7693)</td><td align="left" valign="top">0.8559 (0.7900&#x2010;0.9152)</td><td align="left" valign="top">0.5653 (0.4876&#x2010;0.6349)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L3</td><td align="left" valign="top">0.5093 (0.4594&#x2010;0.5617)</td><td align="left" valign="top">0.8549 (0.7603&#x2010;0.9348)</td><td align="left" valign="top">0.6380 (0.5759&#x2010;0.6942)</td><td align="left" valign="top">0.9602 (0.9539&#x2010;0.9665)</td><td align="left" valign="top">0.7524 (0.6679&#x2010;0.8206)</td><td align="left" valign="top">0.9056 (0.8517&#x2010;0.9489)</td><td align="left" valign="top">0.6158 (0.5498&#x2010;0.6757)</td></tr><tr><td align="left" valign="top" colspan="8">SAG<sup><xref ref-type="table-fn" rid="table5fn3">c</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L1</td><td align="left" valign="top">0.6265 (0.5385&#x2010;0.7084)</td><td align="left" valign="top">0.5423 (0.4130&#x2010;0.6957)</td><td align="left" valign="top">0.5801 (0.4691&#x2010;0.6882)</td><td align="left" valign="top">0.9845 (0.9811&#x2010;0.9885)</td><td align="left" valign="top">0.5565 (0.4338&#x2010;0.6858)</td><td align="left" valign="top">0.7291 (0.6373&#x2010;0.8262)</td><td align="left" valign="top">0.5615 (0.4472&#x2010;0.6730)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L2</td><td align="left" valign="top">0.7247 (0.6723&#x2010;0.7819)</td><td align="left" valign="top">0.8455 (0.7391&#x2010;0.9348)</td><td align="left" valign="top">0.7799 (0.7158&#x2010;0.8381)</td><td align="left" valign="top">0.9845 (0.9811&#x2010;0.9885)</td><td align="left" valign="top">0.8178 (0.7296&#x2010;0.8943)</td><td align="left" valign="top">0.9120 (0.8529&#x2010;0.9607)</td><td align="left" valign="top">0.7684 (0.7016&#x2010;0.8293)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L3</td><td align="left" valign="top">0.7385 (0.6897&#x2010;0.7925)</td><td align="left" valign="top">0.9062 (0.8261&#x2010;0.9783)</td><td align="left" valign="top">0.8134 (0.7600&#x2010;0.8627)</td><td align="left" valign="top">0.9845 (0.9811&#x2010;0.9885)</td><td align="left" valign="top">0.8665 (0.7983&#x2010;0.9259)</td><td align="left" valign="top">0.9443 (0.9012&#x2010;0.9818)</td><td align="left" valign="top">0.8034 (0.7475&#x2010;0.8554)</td></tr><tr><td align="left" valign="top" colspan="8">RAG<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L1</td><td align="left" valign="top">0.5947 (0.5128&#x2010;0.6758)</td><td align="left" valign="top">0.5471 (0.4130&#x2010;0.6957)</td><td align="left" valign="top">0.5687 (0.4578&#x2010;0.6739)</td><td align="left" valign="top">0.9821 (0.9780&#x2010;0.9864)</td><td align="left" valign="top">0.5553 (0.4318&#x2010;0.6809)</td><td align="left" valign="top">0.7315 (0.6369&#x2010;0.8257)</td><td align="left" valign="top">0.5490 (0.4346&#x2010;0.6582)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L2</td><td align="left" valign="top">0.6970 (0.6481&#x2010;0.7586)</td><td align="left" valign="top">0.8524 (0.7391&#x2010;0.9353)</td><td align="left" valign="top">0.7663 (0.7000&#x2010;0.8257)</td><td align="left" valign="top">0.9821 (0.9780&#x2010;0.9864)</td><td align="left" valign="top">0.8156 (0.7295&#x2010;0.8921)</td><td align="left" valign="top">0.9146 (0.8529&#x2010;0.9609)</td><td align="left" valign="top">0.7539 (0.6843&#x2010;0.8159)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>L3</td><td align="left" valign="top">0.7185 (0.6724&#x2010;0.7759)</td><td align="left" valign="top">0.9448 (0.8696&#x2010;1.0000)</td><td align="left" valign="top">0.8159 (0.7736&#x2010;0.8598)</td><td align="left" valign="top">0.9821 (0.9780&#x2010;0.9864)</td><td align="left" valign="top">0.8885 (0.8333&#x2010;0.9350)</td><td align="left" valign="top">0.9631 (0.9251&#x2010;0.9916)</td><td align="left" valign="top">0.8057 (0.7611&#x2010;0.8521)</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup> A total of 1000 random resampling iterations were conducted from 19,983 raw clinical progress notes. The average adverse drug event prevalence was approximately 4.6%. Values are presented as means with percentile-based 95% CIs across 1000 resampling iterations. Prompting strategies evaluated include nonaugmented generation serving as a zero-shot baseline, static-augmented generation integrating static clinical diagnostic rules, and retrieval-augmented generation dynamically injecting top-5 similar cases for context-aware, few-shot prompting. Matching levels are defined as L1 (exact match of sentences, drugs, and reactions), L2 (sentence-level match with drug or reaction discrepancies), and L3 (partial overlap of entities).</p></fn><fn id="table5fn2"><p><sup>b</sup>NAG: nonaugmented generation.</p></fn><fn id="table5fn3"><p><sup>c</sup>SAG: static-augmented generation.</p></fn><fn id="table5fn4"><p><sup>d</sup>RAG: retrieval-augmented generation.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This study investigates the identification of ADEs from clinical course records by synergistically integrating LLMs with RAG technology. Using a progressive research methodology, we demonstrate that the incorporation of a RAG framework substantially enhances the performance of LLMs in the ADE recognition task. Comparative experimental analysis reveals that our proposed RAG solution, which leverages a curated ADE knowledge repository, achieves superior performance across precision, recall, and <italic>F</italic><sub>1</sub>-score metrics, thereby validating its considerable utility in the domain of medical text analysis. Furthermore, this work contributes a novel benchmark corpus, addressing the existing scarcity of annotated data for ADE extraction in the Chinese language.</p><p>In the absence of knowledge augmentation (using NAG prompt), the various LLMs exhibited significant performance disparities in ADE recognition. This variability likely stems from differences in their foundational training corpora, including the breadth of medical data coverage, divergent parameter optimization strategies, and the nuanced depth of their comprehension of Chinese clinical terminology [<xref ref-type="bibr" rid="ref24">24</xref>]. Among the models evaluated, DeepSeek-V3 demonstrated optimal recognition performance under the RAG prompt. DeepSeek-V3, which claimed performance competitive with GPT-4o shortly after its release in January 2025, has rapidly gained global prominence [<xref ref-type="bibr" rid="ref25">25</xref>]. Unlike the proprietary GPT-4o, DeepSeek-V3 is an open-source model that permits local deployment. In a local deployment, its parameter weights can remain insulated from alterations in cloud infrastructure or API updates, thereby supporting long-term consistency and reproducibility while protecting patient data privacy. These advantages create a favorable pathway for the future deployment and external validation of this architectural framework within local health care institutions. However, this study used the cloud API version of DeepSeek-V3 to improve experimental efficiency and standardize batch inference. Local deployment therefore remains a future pathway for privacy-preserving implementation, rather than the deployment mode used in this study.</p><p>In the real-world low-prevalence evaluation, the L3 <italic>F</italic><sub>1</sub>-score differed only marginally between SAG and RAG (0.8134 vs 0.8159), likely because the 2 evaluation settings differed substantially in data distribution. In the same uncurated setting, SAG slightly exceeded RAG in <italic>F</italic><sub>1</sub>-score at L1 (0.5801 vs 0.5687) and L2 (0.7799 vs 0.7663). The curated reference set was relatively balanced and enriched for ADE-positive cases, allowing dynamic retrieval to provide closely matched examples and entity-level contextual cues that improved drug and reaction recognition. By contrast, routine clinical notes had a low ADE prevalence and were dominated by negative records. In this setting, the static clinical rules embedded in SAG already captured common negative contexts, including historical ADE descriptions, prophylactic medication use, and routine monitoring statements. This strong baseline specificity limited the incremental <italic>F</italic><sub>1</sub>-score gain from dynamic retrieval. However, RAG improved recall at the L3 level from 0.9062 to 0.9448, indicating better detection of rare positive ADE cases. Thus, in low-prevalence clinical data, RAG primarily enhances positive-case recall, underscoring its capacity to effectively mitigate the risk of underreporting ADEs.</p><p>Our empirical findings demonstrate that the performance of a conventional, pure language model is markedly inferior to that of a variant augmented with a RAG framework. This outcome aligns with the established theoretical premise that the integration of domain-specific knowledge substantially enhances the capabilities of large models. In contrast to fine-tuning methodologies, RAG obviates the need for voluminous training corpora and protracted training cycles, and circumvents the laborious process of creating contemporary annotated datasets or engaging in repetitive model retraining to assimilate updated, customized knowledge [<xref ref-type="bibr" rid="ref26">26</xref>]. The paramount advantage of the RAG model resides in its profound adaptability [<xref ref-type="bibr" rid="ref19">19</xref>], enabling the continuous assimilation of the latest information on ADE cases&#x2014;a feature of particular salience within the dynamic landscape of medicine and pharmacology, where novel therapeutics are developed at a rapid pace. Consequently, when information regarding an ADE for a newly marketed drug becomes available, it can be seamlessly incorporated into the ADE knowledge repository, facilitating dynamic updates without necessitating extensive modifications to the core model. Furthermore, RAG&#x2019;s capacity to reference and cite antecedent ADE cases enhances the verifiability and perceived relevance of its outputs, thereby effectively mitigating the phenomenon of &#x201C;confabulation,&#x201D; or artifactual generation, which frequently plagues the application of LLMs in medical contexts, while concurrently improving model transparency [<xref ref-type="bibr" rid="ref27">27</xref>]. An error analysis revealed 15 common errors that were uniformly misidentified by all tested models, exposing the inherent limitations of LLMs in processing specific complexities within clinical texts. Critically, we demonstrated that these identification failures could be rectified by supplementing the model&#x2019;s context with pertinent ADE case knowledge via the RAG architecture. This not only furnishes a definitive pathway for augmenting the precision of LLMs in specialized domains but also underscores the intrinsic value of the ADE knowledge repository. Future work should therefore be directed towards the construction of a more comprehensive repository that encompasses not only ADE case reports but also multifaceted information from product information, clinical practice guidelines, and disease manifestations.</p></sec><sec id="s4-2"><title>Comparison With Previous Studies</title><p>To date, systematic investigations into the application of LLMs for general ADE information recognition remain nascent, with most research focusing on narrowly defined ADEs. For instance, Cheligeer et al [<xref ref-type="bibr" rid="ref28">28</xref>] evaluated 4 open-source LLMs in the detection of pulmonary embolism from narrative electronic medical records, while Li et al [<xref ref-type="bibr" rid="ref29">29</xref>] demonstrated the efficacy and robustness of LLMs in extracting postvaccination adverse events from diverse data sources, including the VAERS (US Centers for Disease Control and Prevention and US Food and Drug Administration), Twitter (Twitter, Inc), and Reddit (Reddit, Inc). Our research, in contrast, underscores the broader potential of LLMs to identify all types of ADEs within unstructured Chinese clinical texts. The best-performing RAG-enhanced LLM achieved an <italic>F</italic><sub>1</sub>-score of 96.38%, which is higher than the performance reported in some previous deep learning&#x2013;based ADE extraction studies [<xref ref-type="bibr" rid="ref30">30</xref>]. However, because these studies used different datasets, annotation criteria, and evaluation settings, this comparison should be interpreted only as indirect evidence rather than a head-to-head demonstration of superiority. Nevertheless, our findings suggest that RAG-enhanced LLMs represent a promising framework for ADE extraction, with potential advantages in contextual reasoning, prompt-based adaptation, and flexible knowledge updating [<xref ref-type="bibr" rid="ref31">31</xref>]. Moreover, LLMs obviate the need for extensive labeled datasets and are amenable to continuous refinement as new data becomes available [<xref ref-type="bibr" rid="ref32">32</xref>].</p></sec><sec id="s4-3"><title>Clinical Application</title><p>The methodologies and results presented herein possess considerable potential for translation into clinical practice, where they could elevate the standards of ADE surveillance and reporting and advance pharmacovigilance and drug safety research. Moreover, our approach can be embedded within Clinical Decision Support Systems. Such integration would synergistically combine ADE information with other relevant patient data streams (such as laboratory results, imaging data, and demographic information) to provide clinicians with a more holistic and precise patient assessment, thereby guiding more timely and judicious clinical decision-making. To further enhance the practical utility of our model, we are currently developing a web-based platform, which will feature a user-friendly interface allowing clinicians to query for ADE information within texts through simple conversational input.</p></sec><sec id="s4-4"><title>Limitations and Future Work</title><p>This study has several limitations. First, the knowledge repository was constructed from a single data source, which may result in an incomplete representation of all potential ADEs. Crucially, evaluating models on the same data distribution as the knowledge base risks overfitting and implicit data leakage within the RAG framework, limiting external validity. Multi-institutional validation, using independent data and varied clinical note formats, is therefore essential to verify real-world generalizability. In the future, the integration of multisource information, including pharmaceutical package inserts and expert-derived knowledge, would likely yield a substantial improvement in model performance.</p><p>Second, given the rapid pace of technological iteration in LLMs, the models used in this research may not represent the latest state-of-the-art (eg, the most recent Gemini [Google LLC] or Claude architectures [Anthropic PBC]). While larger-parameter models may offer performance gains, such enhancements must be carefully weighed against the concomitant increase in computational resource expenditure and memory requirements, which presents a significant challenge in resource-constrained settings.</p><p>Third, although the RAG-enhanced LLMs achieved promising performance in ADE identification, this study did not include a direct head-to-head comparison with standard clinical natural language processing baselines, such as fine-tuned clinical Bidirectional Encoder Representations from Transformers (BERT)&#x2013;based models. Therefore, our findings should be interpreted as evidence of the feasibility and potential utility of the proposed framework, rather than definitive evidence of superiority over traditional deep learning approaches. Future work will include direct comparisons with conventional clinical natural language processing models on the same curated dataset and further explore LLM fine-tuning strategies to rigorously evaluate and optimize ADE identification performance.</p><p>Fourth, all clinical notes were strictly deidentified, preventing the extraction and reporting of baseline patient demographics. Despite this limitation, the generalizability of our findings is preserved through the random sampling of heterogeneous data.</p><p>Fifth, although the RAG model achieved high specificity (0.9821) in the real-world evaluation, this corresponds to an approximate false-positive rate of 1.79%. At the observed ADE prevalence of about 4.6%, this would translate into roughly 17 falsely flagged ADE-negative notes per 1000 screened notes, which could increase manual review burden or alert fatigue if deployed without clinician oversight. Future work should calibrate decision thresholds, retrieval filtering, human-in-the-loop review, and iterative knowledge-base expansion to balance sensitivity against false-alert burden while further improving RAG effectiveness and reducing hallucination-related false positives.</p><p>Sixth, all performance metrics in this study were computed at the document level. This evaluation strategy is clinically meaningful for screening whether a note contains ADE information, but document-level precision does not capture entity-level overextraction within otherwise ADE-positive notes. Future work should therefore supplement document-level evaluation to more comprehensively assess extraction granularity and hallucinated entities within positive notes.</p><p>In light of these constraints, our future research will explore the incorporation of knowledge graph embedding techniques [<xref ref-type="bibr" rid="ref33">33</xref>]. By leveraging graph-based structural reasoning, this approach could facilitate the logical inference of ADE knowledge not explicitly contained within the existing database, thereby empowering users to identify potential ADEs with greater accuracy. To further enhance the clinical utility of the proposed system, future investigations should focus on expanding the ADE knowledge base framework and evaluating the system&#x2019;s performance in diverse, real-world clinical settings. To facilitate this translation, we recently developed ADESys [<xref ref-type="bibr" rid="ref34">34</xref>]. Future work will leverage this system for multicenter external validation to enhance our method&#x2019;s robustness and translational value. Such efforts are essential to reducing ADE underreporting and increasing the overall efficiency of pharmacovigilance research.</p></sec><sec id="s4-5"><title>Conclusion</title><p>This study evaluated the effectiveness of a knowledge base-driven RAG framework in identifying ADE information from Chinese clinical narratives. The experimental results across 3 LLMs (DeepSeek-V3, ERNIE 3.5-8K, and GPT-4o) indicate that the RAG strategy consistently improves identification performance compared to NAG and SAG approaches. Furthermore, this research establishes a valuable benchmark, addressing the critical gap in ADE corpora within the Chinese language domain.</p></sec></sec></body><back><ack><p>The authors are grateful to all health administrators and experts who participated in this study for their invaluable support.</p></ack><notes><sec><title>Funding</title><p>This work was funded by the National Natural Science Foundation of China (No. 82474009), Hunan Provincial Natural Science Foundation of China (No. 2023JJ60513 and 2025JJ30025), the Liuzhou City Science and Technology Planning Project (No. 2024YB0103A019), and the Liuzhou Key Laboratory of Clinical Drug Research and Evaluation for Women and Children.</p></sec><sec><title>Data Availability</title><p>The datasets generated or analyzed during this study are available from a GitHub repository [<xref ref-type="bibr" rid="ref23">23</xref>].</p></sec></notes><fn-group><fn fn-type="con"><p>JM and XW contributed equally as the first authors. GY and ML contributed equally as the corresponding authors. JM and XW are responsible for the study design and selection of studies. Data extraction, critical evaluation, and coding were performed by JM, XW, ZF, YK, and ZD. GY and ML contributed to project administration and supervision. JM, XW, GY, and ML drafted the manuscript. All authors critically reviewed and approved the final submitted version of the manuscript.</p><p>GY and ML are co-corresponding authors on this work, and ML can be reached by email at limin@mail.csu.edu.cn</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ADE</term><def><p>adverse drug event</p></def></def-item><def-item><term id="abb2">BERT</term><def><p>Bidirectional Encoder Representations from Transformers</p></def></def-item><def-item><term id="abb3">FDR</term><def><p>false discovery rate</p></def></def-item><def-item><term id="abb4">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb5">NAG</term><def><p>nonaugmented generation</p></def></def-item><def-item><term id="abb6">RAG</term><def><p>retrieval-augmented generation</p></def></def-item><def-item><term id="abb7">SAG</term><def><p>static-augmented generation</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Osanlou</surname><given-names>R</given-names> </name><name name-style="western"><surname>Walker</surname><given-names>L</given-names> </name><name name-style="western"><surname>Hughes</surname><given-names>DA</given-names> </name><name name-style="western"><surname>Burnside</surname><given-names>G</given-names> </name><name name-style="western"><surname>Pirmohamed</surname><given-names>M</given-names> </name></person-group><article-title>Adverse drug reactions, multimorbidity and polypharmacy: a prospective analysis of 1 month of medical admissions</article-title><source>BMJ Open</source><year>2022</year><month>07</month><day>4</day><volume>12</volume><issue>7</issue><fpage>e055551</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2021-055551</pub-id><pub-id pub-id-type="medline">35788071</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Angamo</surname><given-names>MT</given-names> </name><name name-style="western"><surname>Chalmers</surname><given-names>L</given-names> </name><name name-style="western"><surname>Curtain</surname><given-names>CM</given-names> </name><name name-style="western"><surname>Bereznicki</surname><given-names>LRE</given-names> </name></person-group><article-title>Adverse-drug-reaction-related hospitalisations in developed and developing countries: a review of prevalence and contributing factors</article-title><source>Drug Saf</source><year>2016</year><month>09</month><volume>39</volume><issue>9</issue><fpage>847</fpage><lpage>857</lpage><pub-id pub-id-type="doi">10.1007/s40264-016-0444-7</pub-id><pub-id pub-id-type="medline">27449638</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Panagioti</surname><given-names>M</given-names> </name><name name-style="western"><surname>Khan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Keers</surname><given-names>RN</given-names> </name><etal/></person-group><article-title>Prevalence, severity, and nature of preventable patient harm across medical care settings: systematic review and meta-analysis</article-title><source>BMJ</source><year>2019</year><month>07</month><day>17</day><volume>366</volume><fpage>l4185</fpage><pub-id pub-id-type="doi">10.1136/bmj.l4185</pub-id><pub-id pub-id-type="medline">31315828</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wermund</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Haerdtlein</surname><given-names>A</given-names> </name><name name-style="western"><surname>Fehrmann</surname><given-names>W</given-names> </name><name name-style="western"><surname>Weglage</surname><given-names>C</given-names> </name><name name-style="western"><surname>Dreischulte</surname><given-names>T</given-names> </name><name name-style="western"><surname>Jaehde</surname><given-names>U</given-names> </name></person-group><article-title>Drug-event pairs as indicators for the detection of adverse drug reactions during hospitalization in routinely collected electronic data sources</article-title><source>Clin Pharmacol Ther</source><year>2025</year><month>06</month><volume>117</volume><issue>6</issue><fpage>1811</fpage><lpage>1819</lpage><pub-id pub-id-type="doi">10.1002/cpt.3635</pub-id><pub-id pub-id-type="medline">40099752</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Cai</surname><given-names>T</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>S</given-names> </name><etal/></person-group><article-title>High-throughput phenotyping with electronic medical record data using a common semi-supervised approach (PheCAP)</article-title><source>Nat Protoc</source><year>2019</year><month>12</month><volume>14</volume><issue>12</issue><fpage>3426</fpage><lpage>3444</lpage><pub-id pub-id-type="doi">10.1038/s41596-019-0227-6</pub-id><pub-id pub-id-type="medline">31748751</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bucher</surname><given-names>BT</given-names> </name><name name-style="western"><surname>Ferraro</surname><given-names>JP</given-names> </name><name name-style="western"><surname>Finlayson</surname><given-names>SRG</given-names> </name><name name-style="western"><surname>Chapman</surname><given-names>WW</given-names> </name><name name-style="western"><surname>Gundlapalli</surname><given-names>AV</given-names> </name></person-group><article-title>Use of computerized provider order entry events for postoperative complication surveillance</article-title><source>JAMA Surg</source><year>2019</year><month>04</month><day>1</day><volume>154</volume><issue>4</issue><fpage>311</fpage><lpage>318</lpage><pub-id pub-id-type="doi">10.1001/jamasurg.2018.4874</pub-id><pub-id pub-id-type="medline">30586132</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Montastruc</surname><given-names>F</given-names> </name><name name-style="western"><surname>Storck</surname><given-names>W</given-names> </name><name name-style="western"><surname>de Canecaude</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Will artificial intelligence chatbots replace clinical pharmacologists? An exploratory study in clinical practice</article-title><source>Eur J Clin Pharmacol</source><year>2023</year><month>10</month><volume>79</volume><issue>10</issue><fpage>1375</fpage><lpage>1384</lpage><pub-id pub-id-type="doi">10.1007/s00228-023-03547-8</pub-id><pub-id pub-id-type="medline">37566133</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ayers</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Poliak</surname><given-names>A</given-names> </name><name name-style="western"><surname>Dredze</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Comparing physician and artificial intelligence chatbot responses to patient questions posted to a public social media forum</article-title><source>JAMA Intern Med</source><year>2023</year><month>06</month><day>1</day><volume>183</volume><issue>6</issue><fpage>589</fpage><lpage>596</lpage><pub-id pub-id-type="doi">10.1001/jamainternmed.2023.1838</pub-id><pub-id pub-id-type="medline">37115527</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rider</surname><given-names>NL</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chin</surname><given-names>AT</given-names> </name><etal/></person-group><article-title>Evaluating large language model performance to support the diagnosis and management of patients with primary immune disorders</article-title><source>J Allergy Clin Immunol</source><year>2025</year><month>07</month><volume>156</volume><issue>1</issue><fpage>81</fpage><lpage>87</lpage><pub-id pub-id-type="doi">10.1016/j.jaci.2025.02.004</pub-id><pub-id pub-id-type="medline">39956279</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Scaff</surname><given-names>SPS</given-names> </name><name name-style="western"><surname>Reis</surname><given-names>FJJ</given-names> </name><name name-style="western"><surname>Ferreira</surname><given-names>GE</given-names> </name><name name-style="western"><surname>Jacob</surname><given-names>MF</given-names> </name><name name-style="western"><surname>Saragiotto</surname><given-names>BT</given-names> </name></person-group><article-title>Assessing the performance of AI chatbots in answering patients&#x2019; common questions about low back pain</article-title><source>Ann Rheum Dis</source><year>2025</year><month>01</month><volume>84</volume><issue>1</issue><fpage>143</fpage><lpage>149</lpage><pub-id pub-id-type="doi">10.1136/ard-2024-226202</pub-id><pub-id pub-id-type="medline">39874229</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lorenzoni</surname><given-names>G</given-names> </name><name name-style="western"><surname>Gregori</surname><given-names>D</given-names> </name><name name-style="western"><surname>Bressan</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Use of a large language model to identify and classify injuries with free-text emergency department data</article-title><source>JAMA Netw Open</source><year>2024</year><month>05</month><day>1</day><volume>7</volume><issue>5</issue><fpage>e2413208</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.13208</pub-id><pub-id pub-id-type="medline">38805230</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chung</surname><given-names>P</given-names> </name><name name-style="western"><surname>Fong</surname><given-names>CT</given-names> </name><name name-style="western"><surname>Walters</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Aghaeepour</surname><given-names>N</given-names> </name><name name-style="western"><surname>Yetisgen</surname><given-names>M</given-names> </name><name name-style="western"><surname>O&#x2019;Reilly-Shah</surname><given-names>VN</given-names> </name></person-group><article-title>Large language model capabilities in perioperative risk prediction and prognostication</article-title><source>JAMA Surg</source><year>2024</year><month>08</month><day>1</day><volume>159</volume><issue>8</issue><fpage>928</fpage><lpage>937</lpage><pub-id pub-id-type="doi">10.1001/jamasurg.2024.1621</pub-id><pub-id pub-id-type="medline">38837145</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>van Dis</surname><given-names>EAM</given-names> </name><name name-style="western"><surname>Bollen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zuidema</surname><given-names>W</given-names> </name><name name-style="western"><surname>van Rooij</surname><given-names>R</given-names> </name><name name-style="western"><surname>Bockting</surname><given-names>CL</given-names> </name></person-group><article-title>ChatGPT: five priorities for research</article-title><source>Nature</source><year>2023</year><month>02</month><volume>614</volume><issue>7947</issue><fpage>224</fpage><lpage>226</lpage><pub-id pub-id-type="doi">10.1038/d41586-023-00288-7</pub-id><pub-id pub-id-type="medline">36737653</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Azamfirei</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kudchadkar</surname><given-names>SR</given-names> </name><name name-style="western"><surname>Fackler</surname><given-names>J</given-names> </name></person-group><article-title>Large language models and the perils of their hallucinations</article-title><source>Crit Care</source><year>2023</year><month>03</month><day>21</day><volume>27</volume><issue>1</issue><fpage>120</fpage><pub-id pub-id-type="doi">10.1186/s13054-023-04393-x</pub-id><pub-id pub-id-type="medline">36945051</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Williams</surname><given-names>CYK</given-names> </name><name name-style="western"><surname>Bains</surname><given-names>J</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Evaluating large language models for drafting emergency department encounter summaries</article-title><source>PLOS Digit Health</source><year>2025</year><month>06</month><volume>4</volume><issue>6</issue><fpage>e0000899</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000899</pub-id><pub-id pub-id-type="medline">40526634</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xiong</surname><given-names>G</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>A</given-names> </name></person-group><article-title>Improving retrieval-augmented generation in medicine with iterative follow-up questions</article-title><source>Pac Symp Biocomput</source><year>2025</year><volume>30</volume><fpage>199</fpage><lpage>214</lpage><pub-id pub-id-type="doi">10.1142/9789819807024_0015</pub-id><pub-id pub-id-type="medline">39670371</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Tang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Qiao</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>X</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name></person-group><article-title>HW-TSC at TextGraphs-17 Shared Task: Enhancing Inference Capabilities of LLMs with Knowledge Graphs</article-title><year>2024</year><conf-name>Proceedings of TextGraphs-17: Graph-based Methods for Natural Language Processing</conf-name><conf-date>Aug 15, 2024</conf-date><conf-loc>Bangkok, Thailand</conf-loc><publisher-name>Association for Computational Linguistics</publisher-name><pub-id pub-id-type="doi">10.18653/v1/2024.textgraphs-1.11</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lozano</surname><given-names>A</given-names> </name><name name-style="western"><surname>Fleming</surname><given-names>SL</given-names> </name><name name-style="western"><surname>Chiang</surname><given-names>CC</given-names> </name><name name-style="western"><surname>Shah</surname><given-names>N</given-names> </name></person-group><article-title>Clinfo.ai: an open-source retrieval-augmented large language model system for answering medical questions using scientific literature</article-title><source>Pac Symp Biocomput</source><year>2024</year><volume>29</volume><fpage>8</fpage><lpage>23</lpage><pub-id pub-id-type="medline">38160266</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Weinert</surname><given-names>DA</given-names> </name><name name-style="western"><surname>Rauschecker</surname><given-names>AM</given-names> </name></person-group><article-title>Enhancing large language models with retrieval-augmented generation: a radiology-specific approach</article-title><source>Radiol Artif Intell</source><year>2025</year><month>05</month><volume>7</volume><issue>3</issue><fpage>e240313</fpage><pub-id pub-id-type="doi">10.1148/ryai.240313</pub-id><pub-id pub-id-type="medline">40072217</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zakka</surname><given-names>C</given-names> </name><name name-style="western"><surname>Shad</surname><given-names>R</given-names> </name><name name-style="western"><surname>Chaurasia</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Almanac - retrieval-augmented language models for clinical medicine</article-title><source>NEJM AI</source><year>2024</year><month>02</month><volume>1</volume><issue>2</issue><pub-id pub-id-type="doi">10.1056/aioa2300068</pub-id><pub-id pub-id-type="medline">38343631</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ji</surname><given-names>M</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>An</surname><given-names>R</given-names> </name></person-group><article-title>Use of retrieval-augmented large language model for COVID-19 fact-checking: development and usability study</article-title><source>J Med Internet Res</source><year>2025</year><volume>27</volume><fpage>e66098</fpage><pub-id pub-id-type="doi">10.2196/66098</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Qin</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Cui</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Intelligent Chinese patent medicine (CPM) recommendation framework: integrating large language models, retrieval-augmented generation, and the largest CPM dataset</article-title><source>Pharmacol Res</source><year>2025</year><month>09</month><volume>219</volume><fpage>107883</fpage><pub-id pub-id-type="doi">10.1016/j.phrs.2025.107883</pub-id><pub-id pub-id-type="medline">40714300</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="web"><article-title>LLMADE</article-title><source>GitHub</source><access-date>2025-09-15</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/wuxuehong214/LLMADE">https://github.com/wuxuehong214/LLMADE</ext-link></comment></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Spitzl</surname><given-names>D</given-names> </name><name name-style="western"><surname>Mergen</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bauer</surname><given-names>U</given-names> </name><etal/></person-group><article-title>Leveraging large language models for accurate classification of liver lesions from MRI reports</article-title><source>Comput Struct Biotechnol J</source><year>2025</year><volume>27</volume><fpage>2139</fpage><lpage>2146</lpage><pub-id pub-id-type="doi">10.1016/j.csbj.2025.05.019</pub-id><pub-id pub-id-type="medline">40502931</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Conroy</surname><given-names>G</given-names> </name><name name-style="western"><surname>Mallapaty</surname><given-names>S</given-names> </name></person-group><article-title>How China created AI model DeepSeek and shocked the world</article-title><source>Nature</source><year>2025</year><month>02</month><volume>638</volume><issue>8050</issue><fpage>300</fpage><lpage>301</lpage><pub-id pub-id-type="doi">10.1038/d41586-025-00259-0</pub-id><pub-id pub-id-type="medline">39885352</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ke</surname><given-names>YH</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>L</given-names> </name><name name-style="western"><surname>Elangovan</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Retrieval augmented generation for 10 large language models and its generalizability in assessing medical fitness</article-title><source>NPJ Digit Med</source><year>2025</year><month>04</month><day>5</day><volume>8</volume><issue>1</issue><fpage>187</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01519-z</pub-id><pub-id pub-id-type="medline">40185842</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alkaissi</surname><given-names>H</given-names> </name><name name-style="western"><surname>McFarlane</surname><given-names>SI</given-names> </name></person-group><article-title>Artificial hallucinations in ChatGPT: implications in scientific writing</article-title><source>Cureus</source><year>2023</year><month>02</month><volume>15</volume><issue>2</issue><fpage>e35179</fpage><pub-id pub-id-type="doi">10.7759/cureus.35179</pub-id><pub-id pub-id-type="medline">36811129</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cheligeer</surname><given-names>C</given-names> </name><name name-style="western"><surname>Southern</surname><given-names>DA</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Utilizing large language models for detecting hospital-acquired conditions: an empirical study on pulmonary embolism</article-title><source>J Am Med Inform Assoc</source><year>2025</year><month>05</month><day>1</day><volume>32</volume><issue>5</issue><fpage>876</fpage><lpage>884</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocaf048</pub-id><pub-id pub-id-type="medline">40105654</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Viswaroopan</surname><given-names>D</given-names> </name><name name-style="western"><surname>He</surname><given-names>W</given-names> </name><etal/></person-group><article-title>Improving entity recognition using ensembles of deep learning and fine-tuned large language models: a case study on adverse event extraction from VAERS and social media</article-title><source>J Biomed Inform</source><year>2025</year><month>03</month><volume>163</volume><fpage>104789</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2025.104789</pub-id><pub-id pub-id-type="medline">39923968</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Feng</surname><given-names>ZY</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>XH</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>JL</given-names> </name><etal/></person-group><article-title>DKADE: a novel framework based on deep learning and knowledge graph for identifying adverse drug events and related medications</article-title><source>Brief Bioinform</source><year>2023</year><month>07</month><day>20</day><volume>24</volume><issue>4</issue><fpage>bbad228</fpage><pub-id pub-id-type="doi">10.1093/bib/bbad228</pub-id><pub-id pub-id-type="medline">37344167</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bansal</surname><given-names>G</given-names> </name><name name-style="western"><surname>Chamola</surname><given-names>V</given-names> </name><name name-style="western"><surname>Hussain</surname><given-names>A</given-names> </name><name name-style="western"><surname>Guizani</surname><given-names>M</given-names> </name><name name-style="western"><surname>Niyato</surname><given-names>D</given-names> </name></person-group><article-title>Transforming conversations with AI&#x2014;a comprehensive study of ChatGPT</article-title><source>Cogn Comput</source><year>2024</year><month>09</month><volume>16</volume><issue>5</issue><fpage>2487</fpage><lpage>2510</lpage><pub-id pub-id-type="doi">10.1007/s12559-023-10236-2</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Johnson</surname><given-names>D</given-names> </name><name name-style="western"><surname>Goodman</surname><given-names>R</given-names> </name><name name-style="western"><surname>Patrinely</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Assessing the accuracy and reliability of AI-generated medical responses: an evaluation of the Chat-GPT model</article-title><source>Res Sq</source><year>2023</year><month>02</month><day>28</day><fpage>rs.3.rs-2566942</fpage><pub-id pub-id-type="doi">10.21203/rs.3.rs-2566942/v1</pub-id><pub-id pub-id-type="medline">36909565</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xiao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>H</given-names> </name><name name-style="western"><surname>Li</surname><given-names>M</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>R</given-names> </name></person-group><article-title>FuseLinker: leveraging LLM&#x2019;s pre-trained text embeddings and domain knowledge to enhance GNN-based link prediction on biomedical knowledge graphs</article-title><source>J Biomed Inform</source><year>2024</year><month>10</month><volume>158</volume><fpage>104730</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2024.104730</pub-id><pub-id pub-id-type="medline">39326691</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>G</given-names> </name><name name-style="western"><surname>Li</surname><given-names>M</given-names> </name></person-group><article-title>ADESys: a modular system for ADE identification research with LLM-RAG integration</article-title><year>2025</year><conf-name>2025 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</conf-name><conf-date>Dec 15-18, 2025</conf-date><pub-id pub-id-type="doi">10.1109/BIBM66473.2025.11356853</pub-id><pub-id pub-id-type="medline">41859483</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Supplementary figures, including baseline retrieval-augmented generation performance using the full knowledge base, an empirical subanalysis of complex adverse drug event linguistic contexts, the standardized adverse drug event JSON schema and annotation example, the prompt engineering framework, and illustrative examples of L1-L3 recognition levels.</p><media xlink:href="jmir_v28i1e84086_app1.docx" xlink:title="DOCX File, 9691 KB"/></supplementary-material></app-group></back></article>