<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e98580</article-id><article-id pub-id-type="doi">10.2196/98580</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>The Performance of Large Language Models in Extracting Intestinal Symptoms From Electronic Health Records: Retrospective Observational Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Xinyue</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wang</surname><given-names>Quanyu</given-names></name><degrees>MM</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Liu</surname><given-names>Beibei</given-names></name><degrees>MM</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Sang</surname><given-names>Xinyi</given-names></name><degrees>MM</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Wei</surname><given-names>Sheng</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>School of Public Health and Emergency Management, Southern University of Science and Technology</institution><addr-line>1088 Xueyuan Avenue</addr-line><addr-line>Shenzhen</addr-line><addr-line>Guangdong</addr-line><country>China</country></aff><aff id="aff2"><institution>Department of Epidemiology and Biostatistics, Tongji Medical College, School of Public Health, Huazhong University of Science and Technology</institution><addr-line>Wuhan</addr-line><addr-line>Hubei</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Girma</surname><given-names>Abayeneh</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Holgate</surname><given-names>Ben</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Johnson</surname><given-names>Brian</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Sheng Wei, PhD, School of Public Health and Emergency Management, Southern University of Science and Technology, 1088 Xueyuan Avenue, Shenzhen, Guangdong, 518055, China, 86 755-88011926, 86 755-88011926; <email>ws2008cn@gmail.com</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>26</day><month>8</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e98580</elocation-id><history><date date-type="received"><day>16</day><month>04</month><year>2026</year></date><date date-type="rev-recd"><day>04</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>05</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Xinyue Zhang, Quanyu Wang, Beibei Liu, Xinyi Sang, Sheng Wei. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 26.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e98580"/><abstract><sec><title>Background</title><p>Unstructured electronic health records (EHRs) hinder the monitoring of intestinal infections. Large language models (LLMs) enable automated symptom extraction. However, their clinical validation is limited by a lack of systematic multimodel comparisons, unclear prompting strategies, and the privacy risks of cloud-based models (eg, data leakage and cross-border data transfer).</p></sec><sec><title>Objective</title><p>This study aimed to systematically evaluate the performance of locally deployed open-source LLMs across 4 model families in extracting intestinal symptoms from unstructured EHR chief complaints under different prompting strategies.</p></sec><sec sec-type="methods"><title>Methods</title><p>From a citywide health care information platform in Wuhan, China, we randomly selected 1000 chief complaints from outpatient records of intestinal clinics, infectious disease departments, pediatrics, and fever clinics. Six symptoms related to intestinal infectious diseases&#x2014;diarrhea/bloody/mucoid stools, vomiting, abdominal pain, fever, nausea, and rash&#x2014;were manually annotated as a gold-standard dataset. Twelve locally deployed open-source LLMs across 4 families, namely, Gemma3 (1b, 4b, 12b), Qwen3 (1.7b, 8b, 14b), DeepSeek-R1 (1.5b, 7b, 14b), and Llama (Llama2-Chinese 7b, 13b; Llama3.1 8b), were evaluated on the symptom extraction task using the gold-standard dataset. Three prompting strategies (no-role, zero-shot, and few-shot) were tested. Performance metrics included accuracy, precision, recall, <italic>F</italic><sub>1</sub>-score, specificity, balanced accuracy, and inference time. Statistical comparisons used Friedman tests for global differences, followed by Wilcoxon signed-rank and Mann-Whitney <italic>U</italic> tests with Bonferroni and false discovery rate corrections for pairwise comparisons.</p></sec><sec sec-type="results"><title>Results</title><p>Among the 4 families, Qwen3 models showed higher <italic>F</italic><sub>1</sub>-scores and balanced accuracy, with Qwen3-1.7b achieving a macroaveraged <italic>F</italic><sub>1</sub>-score of 0.85 under zero-shot prompting and Qwen3-8b reaching 0.89 under no-role prompting, while Gemma3 demonstrated robust performance at small to medium scales. Symptom-wise, models agreed more on frequent symptoms such as diarrhea and fever, whereas greater variability was observed for rarer symptoms like rash and nausea. The effect of prompting strategy varied across models, with no single strategy consistently outperforming the others. Although some pairwise differences reached statistical significance (<italic>P</italic>&#x003C;.05), the absolute gains in <italic>F</italic><sub>1</sub>-score were small.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>This study provides a systematic comparison of several open-source LLMs on a structured intestinal symptom extraction task. Among the LLM families, Qwen3 models offer a favorable balance between accuracy and efficiency, making them suitable for resource-constrained scenarios.</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>electronic health records</kwd><kwd>intestinal infectious disease</kwd><kwd>surveillance</kwd><kwd>symptom extraction</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Intestinal infectious diseases (IIDs) remain a leading cause of global morbidity and mortality. Enteric infectious diseases cause substantial morbidity and mortality, disproportionately affecting children younger than 5 years of age [<xref ref-type="bibr" rid="ref1">1</xref>]. The rapid transmission dynamics through contaminated food, water, and fomites render these infections particularly susceptible to explosive outbreaks, underscoring the imperative for robust surveillance systems capable of early detection and rapid response.</p><p>Syndromic surveillance serves as the critical sentinel for early epidemic warning. Electronic health records (EHRs) contain vast amounts of unstructured symptom data in chief complaints [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>], representing the foundational data source for surveillance [<xref ref-type="bibr" rid="ref4">4</xref>]. However, conventional manual extraction methods are limited by inefficiency and subjective biases [<xref ref-type="bibr" rid="ref5">5</xref>]. Traditional natural language processing approaches for symptom extraction mainly include rule-based systems and supervised learning models. Rule-based methods rely on manually constructed regular expressions and keyword dictionaries. They are interpretable but poorly generalizable, failing to cover diverse colloquial expressions or to parse negation and conditional clauses [<xref ref-type="bibr" rid="ref6">6</xref>]. Supervised learning models partially alleviate the generalization issue but require large expert-annotated corpora, making them costly and slow to adapt to new or rare diseases [<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref9">9</xref>]. Despite these attempts, primary health care institutions still lack efficient and reliable automated screening tools. Consequently, the development of automated symptom extraction tools capable of delivering standardized, high-throughput screening has emerged as an imperative for enhancing prevention capabilities and enabling real-time public health intelligence [<xref ref-type="bibr" rid="ref10">10</xref>].</p><p>Large language models (LLMs) offer a transformative solution to the limitations of traditional medical natural language processing. Through engineered prompts, LLMs such as Qwen [<xref ref-type="bibr" rid="ref11">11</xref>] and Llama [<xref ref-type="bibr" rid="ref12">12</xref>] can convert unstructured complaints into standardized symptom labels without escalating annotation costs, addressing critical gaps in rare disease recognition and early warning scenarios. In this study, we locally deployed open-source models, including Qwen3, DeepSeek-R1, Gemma3, Llama2-Chinese, and Llama3.1, which were adopted in Chinese clinical tasks and supported reliable local deployment. This privacy-preserving architecture is particularly critical in jurisdictions with stringent data protection regulations, where the export of personal health data to remote servers is legally prohibited [<xref ref-type="bibr" rid="ref13">13</xref>].</p><p>This study aimed to systematically evaluate the performance of 12 locally deployed open-source LLMs across 4 model families in extracting intestinal symptoms from unstructured EHR chief complaints using no-role, zero-shot, and few-shot prompting without task-specific fine-tuning. By benchmarking diverse models and prompt strategies, we aimed to identify optimal configurations that balance timeliness with accuracy, thereby providing scientific evidence for IID symptom monitoring.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>We conducted a retrospective study using routinely collected EHRs from a citywide health care information platform in Wuhan, China. Outpatient records from January 1, 2023, to December 31, 2025, were included. From the eligible records, 1000 chief complaints were randomly selected to construct a gold-standard dataset through manual annotation of intestinal infection symptoms. Twelve locally deployed open-source LLMs were then evaluated under 3 prompting strategies (no-role, zero-shot, and few-shot) to assess symptom extraction performance. The study followed the STARD-AI (Standards for Reporting Diagnostic Accuracy Studies&#x2013;AI) reporting guideline for diagnostic accuracy studies (<xref ref-type="supplementary-material" rid="app6">Checklist 1</xref>).</p></sec><sec id="s2-2"><title>Data Sources</title><p>This study used data from the citywide health care information platform in Wuhan, which enables interoperability of health care data across all medical institutions within the city, including hospitals, community health centers, and specialized clinics. The platform captures health care records spanning outpatient visits, emergency encounters, inpatient admissions, and health screening examinations, encompassing EHRs, laboratory test results, and medication prescriptions.</p><p>A total of 167,567,887 deidentified outpatient records were obtained from the health care information platform from January 1, 2023, to December 31, 2025. All EHRs for the full calendar years were complete. Each record contained an anonymized patient identifier, clinical department, encounter timestamp, and chief complaint. Only the chief complaint field was provided to the LLMs; other fields (eg, department and timestamp) and any additional clinical notes (eg, physical examination and diagnosis) were not used for symptom extraction. An example of a chief complaint (translated from Chinese) was &#x201C;fever for 2 days, diarrhea three times per day with sticky stool.&#x201D;</p><p>All records were deidentified before any analysis. Regular expression matching was used to detect patterns of personally identifiable information, including patient names, phone numbers, and home addresses, and detected identifiers were replaced with masked placeholders. The unique patient identifiers were retained only as linkage keys for deduplication and were never input into the LLMs.</p><p>Given the large volume of chief complaint data in outpatient records, we initially filtered records using keyword-based filtering for terms potentially related to intestinal symptoms. Records with missing values or containing invalid entries such as &#x201C;not filled,&#x201D; &#x201C;chief complaint unclear,&#x201D; or &#x201C;not specified&#x201D; were excluded. Duplicate entries were then removed based on a unique identifier, onset time, and chief complaint. From the filtered dataset, we performed simple random sampling without replacement to select 1000 chief complaints for manual annotation and model evaluation. This sample size was determined by the practical constraints of manual annotation. Detailed preprocessing counts are provided in the flow diagram in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-3"><title>Symptoms Related to IIDs</title><p>To identify clinical symptoms associated with IIDs, we conducted a comprehensive review of authoritative guidelines, infectious disease textbooks, and relevant studies, focusing on viral, bacterial, and parasitic enteric infections. Symptoms were initially categorized as systemic and gastrointestinal manifestations based on established case definitions (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). To expand and validate the symptom inventory, we conducted a systematic literature search in CNKI, Wanfang Data, PubMed, and Web of Science for papers published before June 30, 2025, using combinations of terms, including &#x201C;intestinal infectious disease,&#x201D; &#x201C;enteric infection,&#x201D; and &#x201C;infectious diarrhea&#x201D; paired with &#x201C;syndromic surveillance.&#x201D; From the retrieved literature, we compiled a database of symptoms, surveillance systems, and monitoring timelines.</p></sec><sec id="s2-4"><title>Symptom Extraction Using Locally Deployed LLMs</title><p>Twelve open-source models from 5 series were selected: Meta&#x2019;s Llama-2 and Llama3.1; DeepSeek-R1; Alibaba&#x2019;s Qwen3; and Google&#x2019;s Gemma3, with parameter scales ranging from 1.7 billion to 14 billion. These families represent some of the most widely adopted and actively maintained open-source models. They include grouped-query attention with architectural refinements (Llama2-Chinese and Llama3.1), mixture-of-experts with multihead latent attention (DeepSeek-R1), native Chinese optimization with controllable reasoning modes (Qwen3), and sliding-window attention for multilingual efficiency (Gemma3). The selected models also cover a broad spectrum of Chinese language capability, ranging from natively Chinese-trained models to posttrained adaptations and multilingual baselines [<xref ref-type="bibr" rid="ref14">14</xref>-<xref ref-type="bibr" rid="ref16">16</xref>]. All are compatible with local inference frameworks, ensuring compliance with rigorous data privacy standards essential for processing EHRs in public health settings [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>]. The selection of LLMs also reflected the practical constraints of the deployment environment. The health care platform operated on an isolated internal network with no internet access and limited graphical processing unit (GPU) resources, favoring locally deployable, quantized models.</p><p>To ensure data security, all models were deployed locally using Ollama on an Intel Xeon W-2255 processor equipped with an NVIDIA RTX A6000 GPU (48 GB VRAM) running Windows 11, version 24H2 (build 26100.4061). All models were run in standard nonthinking mode using 4-bit-quantized checkpoints (GGUF format). Inference time was measured as the end-to-end wall-clock time from sending the request to the API until the complete response was received, covering both prompt processing and output generation. All models were evaluated under identical hardware and software settings to ensure fair comparisons.</p><p>The dataset was constructed from outpatient EHRs. To evaluate symptom extraction by LLMs, we constructed a dedicated evaluation dataset. As symptoms of IIDs are mainly documented in chief complaints and cases are primarily managed in intestinal clinics, infectious disease departments, pediatrics, and fever clinics, we limited sampling to these departments to ensure coverage of target symptoms. We then randomly selected 1000 eligible records to form the model dataset without additional symptom stratification, preserving the natural distribution of symptoms in the original clinical population. Symptom labels were ascertained by 2 independent raters, both of whom held a master&#x2019;s degree in public health and received training in clinical symptom annotation. Symptom-specific Cohen &#x03BA; ranged from 0.87 to 1.00. The overall mean &#x03BA; was 0.92 (SD 0.02). Discrepancies were resolved through consensus adjudication to establish the reference standard. Full symptom-specific &#x03BA; values are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Annotations were completed before model inference was performed.</p></sec><sec id="s2-5"><title>Prompt Engineering</title><p>Three structured prompting strategies were evaluated: no-role prompting is defined as the baseline prompting without role definition; zero-shot prompting is defined as task instructions and output formatting without exemplars; few-shot prompting is defined as task guidance with curated input-output examples. To refine these prompts, we used a separate development set of 50 chief complaints randomly selected from the same data source, which were excluded from the final 1000-sample evaluation set. Refinement focused on ensuring output format compliance and parser robustness, without altering symptom classification criteria.</p><p>Each prompt instructed the model to return a JSON object with a predefined schema containing the symptom list. After receiving the raw text response, a rule-based parser extracted the first valid JSON object or array, validated the symptom list, and collected the symptom codes. If no parsable structure was found, the parser returned an empty list, which was treated as &#x201C;no symptoms extracted.&#x201D; The codes were then converted into binary indicator columns (one per symptom). Although models occasionally deviated from the required format (eg, adding extra text before or after the JSON), the parser tolerated such variations by focusing on the JSON structure. Prompt refinement procedures are provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. Batch inference was performed with fixed hyperparameters (temperature at 0.2, top-k=10, and top-p=0.5) across all models. This temperature setting was chosen to ensure a consistent comparison while allowing flexibility for nonstandard symptom expressions, consistent with prior work on clinical text mining [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>].</p></sec><sec id="s2-6"><title>Statistical Analysis</title><p>Statistical analyses were performed to quantify performance differences across model configurations, identify key determinants of performance, and account for the hierarchical structure of the data. Model performance was evaluated at the individual symptom level relative to gold-standard manual annotations. For each of the 6 symptoms, standard binary classification metrics were computed from the confusion matrix: accuracy, precision, recall (sensitivity), specificity, balanced accuracy, and <italic>F</italic><sub>1</sub>-score. To provide an overall summary of performance across symptoms, a macroaveraged <italic>F</italic><sub>1</sub>-score was calculated as the unweighted mean of the 6 symptom-specific <italic>F</italic><sub>1</sub>-scores. Computational efficiency was quantified as the average inference time per sample (in seconds per chief complaint), derived from total runtime logs for each model-prompt configuration.</p><p>Differences in performance metrics across prompting strategies (no-role, zero-shot, and few-shot) and parameter-size groups were assessed using the Friedman test on aligned symptom-level data. For significant omnibus tests (&#x03B1;&#x003C;0.05), post hoc pairwise comparisons were performed using the Dunn test with Benjamini-Hochberg false discovery rate (FDR) correction to control for multiple testing. For inference time, the Kruskal-Wallis <italic>H</italic> test was used to compare parameter-size groups, with pairwise follow-up via Mann-Whitney <italic>U</italic> tests and FDR correction.</p><p>To disentangle the independent and interactive effects of parameter size and prompting strategy, while accounting for the hierarchical data structure (multiple symptoms nested within each model-prompt combination), linear mixed models (LMMs) were fit for each performance metric. The fixed-effects specification was defined as follows:</p><disp-formula id="E1"><mml:math id="eqn1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mi>Z</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msub><mml:mo>&#x22C5;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>Z</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mn>5</mml:mn></mml:mrow></mml:msub><mml:mo>&#x22C5;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mi>&#x03BD;</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>&#x03F5;</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <italic>Y<sub>ij</sub></italic> is the performance metric for the <italic>i</italic>th observation within the <italic>j</italic>th symptom; <italic>&#x03B2;</italic><sub>0</sub> is the overall intercept; <italic>Z<sub>ij</sub></italic> and <italic>F<sub>ij</sub></italic> are indicator variables for prompting strategy (zero-shot and few-shot, with no-role as reference); <italic>S<sub>j</sub></italic> is the numerical parameter size in billions; <italic>&#x03B2;</italic><sub>1</sub> to <italic>&#x03B2;</italic><sub>5</sub> are the fixed-effect coefficients for the main effects and their interaction; <italic>&#x03C5;</italic><sub><italic>j</italic></sub>~<italic>N</italic>(0, &#x03C3;<sup>2</sup><sub>&#x03C5;</sub>) is a random intercept for each symptom, capturing symptom-level variation; <italic>&#x03B5;</italic><sub><italic>ij</italic></sub>~<italic>N</italic>(0, &#x03C3;<sup>2</sup><sub>&#x03C5;</sub>) is the residual error.</p><p>Models were fit using restricted maximum likelihood. The significance of fixed effects was assessed using Wald tests. All statistical analyses and visualizations were performed using Python software (version 3.12.7). Two-tailed tests were used with a significance level of &#x03B1;=.05.</p></sec><sec id="s2-7"><title>Ethical Considerations</title><p>This retrospective study used deidentified EHRs from a citywide health care information platform in Wuhan, China, and did not involve human interventions or the collection of sensitive personal information. The requirement for informed consent was waived because all data were anonymized prior to analysis. All analyses were conducted on a secure institutional server; only aggregate statistics were reported, and no patient identifiers appeared in any figures, tables, or appendices. The study was conducted in accordance with the Declaration of Helsinki and the Measures for the Ethical Review of Life Sciences and Medical Research Involving Humans (2023, China), under which ethical review is waived for research using anonymized data that does not involve human participants or sensitive personal information.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Symptoms Identified</title><p>We selected studies focused on IID symptom monitoring and extracted information on study titles, symptoms, surveillance systems, and study timelines. In total, we included 172 related papers in the final review (<xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>). Drawing from surveillance programs, treatment guidelines, expert consensus, and research findings, we selected 6 representative core symptoms for IIDs. These included 3 gastrointestinal symptoms (diarrhea/bloody/mucoid stools, vomiting, and abdominal pain) and 3 systemic symptoms (fever, nausea, and rash), the details of which are provided in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p></sec><sec id="s3-2"><title>Overall Performance and Efficiency</title><p><xref ref-type="table" rid="table1">Table 1</xref> summarizes global performance across all model-prompt configurations, presenting descriptive statistics, including macro <italic>F</italic><sub>1</sub>-score, balanced accuracy, and inference time. Additional metrics are provided in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>. The results show variation in performance across different models, parameter sizes, and prompting strategies. Among all configurations, Qwen3-8b with no-role prompting achieved a macro <italic>F</italic><sub>1</sub>-score of 0.89 and balanced accuracy of 0.95&#x00B1;0.06, with an inference time of 3.64 seconds per record, whereas Qwen3-1.7b under zero-shot prompting achieved a macro <italic>F</italic><sub>1</sub>-score of 0.85 with a faster inference time of 1.61 seconds. By comparison, Llama3.1-8b achieved macro <italic>F</italic><sub>1</sub>-scores of 0.35 to 0.40 across prompting strategies, with substantially longer inference times (4.29-11.74 s). Inference time also varied substantially, with larger models generally requiring more processing time per record. This aligns with expectations for the computational resource requirements of larger LLM configurations.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Macro <italic>F</italic><sub>1</sub>-score, balanced accuracy, and inference time for each model and prompting strategy.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model and prompt</td><td align="left" valign="bottom">Macro <italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">Balanced accuracy, mean (SD)</td><td align="left" valign="bottom">Time (seconds/record)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="4">Qwen3-1.7b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No-role</td><td align="char" char="." valign="top">0.63</td><td align="char" char="." valign="top">0.85 (0.13)</td><td align="char" char="." valign="top">1.40</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zero-shot</td><td align="left" valign="top">0.85</td><td align="left" valign="top">0.91 (0.09)</td><td align="left" valign="top">1.61</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Few-shot</td><td align="left" valign="top">0.83</td><td align="left" valign="top">0.91 (0.08)</td><td align="left" valign="top">1.52</td></tr><tr><td align="left" valign="top" colspan="4">Qwen3-8b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No-role</td><td align="char" char="." valign="top">0.89</td><td align="char" char="." valign="top">0.95 (0.06)</td><td align="char" char="." valign="top">3.64</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zero-shot</td><td align="left" valign="top">0.78</td><td align="left" valign="top">0.85 (0.11)</td><td align="left" valign="top">6.06</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Few-shot</td><td align="left" valign="top">0.82</td><td align="left" valign="top">0.88 (0.09)</td><td align="left" valign="top">8.04</td></tr><tr><td align="left" valign="top" colspan="4">Qwen3-14b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No-role</td><td align="left" valign="top">0.85</td><td align="left" valign="top">0.92 (0.09)</td><td align="left" valign="top">12.80</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zero-shot</td><td align="left" valign="top">0.82</td><td align="left" valign="top">0.89 (0.09)</td><td align="left" valign="top">10.13</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Few-shot</td><td align="left" valign="top">0.87</td><td align="left" valign="top">0.91 (0.08)</td><td align="left" valign="top">11.23</td></tr><tr><td align="left" valign="top" colspan="4">DeepSeek-R1-1.5b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No-role</td><td align="char" char="." valign="top">0.11</td><td align="char" char="." valign="top">0.53 (0.05)</td><td align="char" char="." valign="top">2.84</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zero-shot</td><td align="left" valign="top">0.32</td><td align="left" valign="top">0.76 (0.13)</td><td align="left" valign="top">2.91</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Few-shot</td><td align="left" valign="top">0.36</td><td align="left" valign="top">0.73 (0.12)</td><td align="left" valign="top">2.75</td></tr><tr><td align="left" valign="top" colspan="4">DeepSeek-R1-7b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No-role</td><td align="char" char="." valign="top">0.40</td><td align="char" char="." valign="top">0.68 (0.18)</td><td align="char" char="." valign="top">3.34</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zero-shot</td><td align="left" valign="top">0.74</td><td align="left" valign="top">0.87 (0.10)</td><td align="left" valign="top">3.17</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Few-shot</td><td align="left" valign="top">0.70</td><td align="left" valign="top">0.90 (0.10)</td><td align="left" valign="top">2.55</td></tr><tr><td align="left" valign="top" colspan="4">DeepSeek-R1-14b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No-role</td><td align="char" char="." valign="top">0.63</td><td align="char" char="." valign="top">0.79 (0.19)</td><td align="char" char="." valign="top">8.32</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zero-shot</td><td align="left" valign="top">0.70</td><td align="left" valign="top">0.94 (0.08)</td><td align="left" valign="top">7.05</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Few-shot</td><td align="left" valign="top">0.73</td><td align="left" valign="top">0.94 (0.08)</td><td align="left" valign="top">19.83</td></tr><tr><td align="left" valign="top" colspan="4">Gemma3-1b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No-role</td><td align="char" char="." valign="top">0.51</td><td align="char" char="." valign="top">0.81 (0.10)</td><td align="char" char="." valign="top">0.20</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zero-shot</td><td align="left" valign="top">0.19</td><td align="left" valign="top">0.61 (0.08)</td><td align="left" valign="top">0.69</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Few-shot</td><td align="left" valign="top">0.27</td><td align="left" valign="top">0.68 (0.11)</td><td align="left" valign="top">0.63</td></tr><tr><td align="left" valign="top" colspan="4">Gemma3-4b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No-role</td><td align="char" char="." valign="top">0.51</td><td align="char" char="." valign="top">0.81 (0.10)</td><td align="char" char="." valign="top">0.66</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zero-shot</td><td align="left" valign="top">0.55</td><td align="left" valign="top">0.92 (0.06)</td><td align="left" valign="top">0.61</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Few-shot</td><td align="left" valign="top">0.58</td><td align="left" valign="top">0.86 (0.08)</td><td align="left" valign="top">0.55</td></tr><tr><td align="left" valign="top" colspan="4">Gemma3-12b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No-role</td><td align="char" char="." valign="top">0.77</td><td align="char" char="." valign="top">0.96 (0.04)</td><td align="char" char="." valign="top">0.32</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zero-shot</td><td align="left" valign="top">0.63</td><td align="left" valign="top">0.95 (0.05)</td><td align="left" valign="top">1.07</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Few-shot</td><td align="left" valign="top">0.77</td><td align="left" valign="top">0.94 (0.06)</td><td align="left" valign="top">1.13</td></tr><tr><td align="left" valign="top" colspan="4">Llama2-Chinese-7b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No-role</td><td align="char" char="." valign="top">0.00</td><td align="char" char="." valign="top">0.50 (0.00)</td><td align="char" char="." valign="top">13.91</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zero-shot</td><td align="left" valign="top">0.03</td><td align="left" valign="top">0.50 (0.01)</td><td align="left" valign="top">2.68</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Few-shot</td><td align="left" valign="top">0.11</td><td align="left" valign="top">0.57 (0.08)</td><td align="left" valign="top">6.04</td></tr><tr><td align="left" valign="top" colspan="4">Llama2-Chinese-13b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No-role</td><td align="char" char="." valign="top">0.00</td><td align="char" char="." valign="top">0.50 (0.00)</td><td align="char" char="." valign="top">7.14</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zero-shot</td><td align="left" valign="top">0.24</td><td align="left" valign="top">0.65 (0.09)</td><td align="left" valign="top">4.88</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Few-shot</td><td align="left" valign="top">0.27</td><td align="left" valign="top">0.78 (0.12)</td><td align="left" valign="top">1.74</td></tr><tr><td align="left" valign="top" colspan="4">Llama3-8b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No-role</td><td align="char" char="." valign="top">0.35</td><td align="char" char="." valign="top">0.88 (0.07)</td><td align="char" char="." valign="top">4.29</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zero-shot</td><td align="left" valign="top">0.40</td><td align="left" valign="top">0.91 (0.06)</td><td align="left" valign="top">7.20</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Few-shot</td><td align="left" valign="top">0.37</td><td align="left" valign="top">0.90 (0.06)</td><td align="left" valign="top">11.74</td></tr></tbody></table></table-wrap></sec><sec id="s3-3"><title>Performance and Inference Time by Model</title><p>The relationship between model scale, performance, and computational cost across all configurations is summarized in <xref ref-type="fig" rid="figure1">Figure 1</xref>. Violin plots of symptom-level macro <italic>F</italic><sub>1</sub>-scores (<xref ref-type="fig" rid="figure1">Figure 1A</xref>) and balanced accuracy (<xref ref-type="fig" rid="figure1">Figure 1B</xref>) reveal different distribution patterns across model families. Qwen3 models, especially the 8b and 14b variants, exhibited high and narrow distributions for both metrics, indicating stable performance. Llama2-Chinese models exhibited lower distributions. Llama3.1-8b showed modestly higher and slightly narrower distributions than Llama2-Chinese, but remained below Qwen3 models. Inference time increased nonlinearly with parameter size and prompting strategy (<xref ref-type="fig" rid="figure1">Figure 1C</xref>). Among the largest models, Qwen3-14b required approximately 12.80 seconds per record under no-role prompting, while DeepSeek-R1-14b required 19.83 seconds under few-shot prompting, the longest latency observed. Gemma3 models maintained consistently short inference times regardless of scale or prompting condition. The efficiency-accuracy trade-off is further illustrated in the scatter plot (<xref ref-type="fig" rid="figure1">Figure 1D</xref>). Gemma3 models cluster toward the left side, reflecting faster inference times, whereas Qwen3 models occupy higher positions, indicating higher <italic>F</italic><sub>1</sub>-scores. DeepSeek-R1 models were more dispersed, with smaller variants falling in the mid-range and the 14b few-shot setting achieving higher accuracy at the cost of longer inference. Llama3.1-8b appears in the lower-right region, reflecting its moderate <italic>F</italic><sub>1</sub>-scores and long inference times, while Llama2-Chinese models occupy the lower-left region with low <italic>F</italic><sub>1</sub>-scores and relatively short inference times. The performance ranking, integrating normalized <italic>F</italic><sub>1</sub>-scores with processing times, is detailed in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Performance and efficiency of large language models for symptom extraction as a function of parameter size. (A) <italic>F</italic><sub>1</sub>-score distribution across all symptoms and prompts for each model. (B) Balanced accuracy distribution across all symptoms and prompts for each model. (C) Inference time (seconds per 1000 chief complaints) for each model under 3 prompting strategies: no-role (baseline instruction), zero-shot (task description without examples), and few-shot (with input-output exemplars). (D) <italic>F</italic><sub>1</sub>-score vs inference time; each point represents one model-prompt combination. Points are colored by model family and series (Qwen3, DeepSeek-R1, Gemma3, Llama2-Chinese, and Llama3.1). Model families are arranged from left to right in order of increasing parameter size within each family.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e98580_fig01.png"/></fig></sec><sec id="s3-4"><title>Performance Across Prompting Strategies Within Models</title><p><italic>F</italic><sub>1</sub>-scores across the 3 prompting strategies for each model are shown in <xref ref-type="fig" rid="figure2">Figure 2</xref>. The effect of prompting strategy was model-dependent, and no single approach dominated across all models. Qwen3-8b achieved its highest <italic>F</italic><sub>1</sub>-score of 0.89 with no-role prompting, while its few-shot and zero-shot scores were 0.82 and 0.78, respectively. DeepSeek-R1-7b performed best with zero-shot prompting (<italic>F</italic><sub>1</sub>-score=0.74), and DeepSeek-R1-14b maintained relatively stable performance across all 3 prompt types. Llama3.1-8b achieved its highest <italic>F</italic><sub>1</sub>-score of 0.40 under zero-shot prompting, while no-role and few-shot exhibited lower or similar scores. For models with low baseline performance (eg, Llama2-Chinese-7b), none of the prompts improved the <italic>F</italic><sub>1</sub>-score beyond 0.11. These findings indicated that the optimal prompting strategy varied by model family and that relying solely on prompt engineering cannot reliably guarantee improved performance, while model architecture and parameter scale appeared to exert an influence on symptom extraction accuracy.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Impact of prompting strategy on <italic>F</italic><sub>1</sub>-score across individual models. Box plots show the <italic>F</italic><sub>1</sub>-score for 3 prompting strategies (no-role, zero-shot, and few-shot) for each model. Significant differences between prompt pairs are indicated by lines above the boxes: *<italic>P</italic>&#x003C;.05, **<italic>P</italic>&#x003C;.01, ***<italic>P</italic>&#x003C;.001. Nonsignificant comparisons are not marked.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e98580_fig02.png"/></fig></sec><sec id="s3-5"><title>Comparison of Key Configurations</title><p>Overall, our results revealed a significant effect of prompting strategy on most performance metrics (<italic>F</italic><sub>1</sub>-score: <italic>&#x03C7;</italic><sup>2</sup><sub>2</sub>=163.9; <italic>P</italic>&#x003C;.001; balanced accuracy: <italic>&#x03C7;</italic><sup>2</sup><sub>2</sub>=151.8; <italic>P</italic>&#x003C;.001). When averaged across symptoms (macro <italic>F</italic><sub>1</sub>-scores), the differences between prompting strategies were small and inconsistent (<xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>). Pairwise comparisons (<xref ref-type="table" rid="table2">Table 2</xref>) showed that the larger-size parameter groups were slower than smaller parameter groups (FDR-adjusted <italic>P</italic>&#x003C;.05). Within the DeepSeek-R1 family, the 14b model significantly outperformed the 1.5b model in accuracy and specificity under both no-role and zero-shot prompts. In the Gemma3 family, the 12b model surpassed the 1b model across multiple metrics, particularly under few-shot prompting. For Llama2-Chinese, the 13b model showed better recall and balanced accuracy than the 7b model under few-shot and zero-shot prompts. Detailed pairwise results are provided in <xref ref-type="table" rid="table2">Table 2</xref>.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Significant pairwise comparisons of prompting strategies and parameter sizes with respect to performance<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Family: comparison models</td><td align="left" valign="bottom">Metric</td><td align="left" valign="bottom">Direction</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="3">DeepSeek-R1: 1.5b vs 14b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No-role</td><td align="left" valign="top">Accuracy</td><td align="char" char="." valign="top">14b&#x003E;1.5b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No-role</td><td align="left" valign="top">Specificity</td><td align="left" valign="top">14b&#x003E;1.5b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zero-shot</td><td align="left" valign="top">Accuracy</td><td align="left" valign="top">14b&#x003E;1.5b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zero-shot</td><td align="left" valign="top">Specificity</td><td align="left" valign="top">14b&#x003E;1.5b</td></tr><tr><td align="left" valign="top" colspan="3">Gemma3: 1b vs 12b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Few-shot</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td><td align="char" char="." valign="top">12b&#x003E;1b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Few-shot</td><td align="left" valign="top">Balanced accuracy</td><td align="left" valign="top">12b&#x003E;1b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Few-shot</td><td align="left" valign="top">Accuracy</td><td align="left" valign="top">12b&#x003E;1b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Few-shot</td><td align="left" valign="top">Specificity</td><td align="left" valign="top">12b&#x003E;1b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No-role</td><td align="left" valign="top">Balanced accuracy</td><td align="left" valign="top">12b&#x003E;1b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No-role</td><td align="left" valign="top">Recall</td><td align="left" valign="top">12b&#x003E;1b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zero-shot</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">12b&#x003E;1b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zero-shot</td><td align="left" valign="top">Balanced accuracy</td><td align="left" valign="top">12b&#x003E;1b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zero-shot</td><td align="left" valign="top">Accuracy</td><td align="left" valign="top">12b&#x003E;1b</td></tr><tr><td align="left" valign="top" colspan="3">Llama2-Chinese: 7b vs 13b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Few-shot</td><td align="left" valign="top">Recall</td><td align="char" char="." valign="top">13b&#x003E;7b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zero-shot</td><td align="left" valign="top">Balanced accuracy</td><td align="left" valign="top">13b&#x003E;7b</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Zero-shot</td><td align="left" valign="top">Recall</td><td align="left" valign="top">13b&#x003E;7b</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Only comparisons with FDR-adjusted <italic>P</italic>&#x003C;.05 are shown. The Mann-Whitney <italic>U</italic> test was applied to each contrast. Within-family comparisons used the smallest and largest parameter sizes available within that family (DeepSeek-R1: 1.5b vs 14b; Gemma3: 1b vs 12b; Llama2-Chinese: 7b vs 13b).</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-6"><title>Symptom-Specific Performance</title><p><xref ref-type="fig" rid="figure3">Figure 3</xref> shows <italic>F</italic><sub>1</sub>-scores for each of the 6 symptoms across models and prompting strategies. The most challenging symptom was nausea, with <italic>F</italic><sub>1</sub>-scores below 0.5 for all models except Qwen3-8b and Qwen3-1.7b under few-shot prompting. Diarrhea was identified with high accuracy (<italic>F</italic><sub>1</sub>-score&#x003E;0.8) by most models. The few-shot prompting strategy improved recognition of less frequent symptoms (eg, rash), whereas the no-role strategy often failed to detect them (<italic>F</italic><sub>1</sub>-score&#x003C;0.1 for the Llama2-Chinese family). These patterns suggest that few-shot examples help the model generalize to underrepresented symptom classes. Confusion matrices for all models across 6 target symptoms are detailed in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Heatmaps of <italic>F</italic><sub>1</sub>-score for individual symptoms. Each panel corresponds to 1 of the 6 symptoms: (A) diarrhea, (B) vomiting, (C) abdominal pain, (D) fever, (E) nausea, and (F) rash. Color intensity reflects the <italic>F</italic><sub>1</sub>-score, ranging from 0 to 1. Models are ordered by family and increasing parameter size. The heatmaps reveal substantial variation in symptom-level performance across models and prompting strategies.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e98580_fig03.png"/></fig></sec><sec id="s3-7"><title>Effect of Parameter Size on Performance Metrics</title><p>Linear mixed-model analysis showed the effect of parameter size on performance metrics, as illustrated in <xref ref-type="fig" rid="figure4">Figure 4</xref>. Under few-shot prompting, each 1 billion increase in parameter size was associated with an increase of 0.0102 in balanced accuracy (95% CI 0.0023&#x2010;0.0180). Similarly, recall improved by 0.0189 per 1 billion (95% CI 0.0034&#x2010;0.0344). Trends toward improvement were observed for precision, though this did not reach conventional significance (&#x03B2;=0.0146, 95% CI &#x2212;0.0015 to 0.0307; &#x03B2;=0.0075, 95% CI &#x2212;0.0007 to 0.0157). Other metrics (accuracy, <italic>F</italic><sub>1</sub>-score, and specificity) did not show a statistically significant association with parameter size (all <italic>P</italic>&#x003E;.05). Full model results, including coefficients and 95% CIs, are provided in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Effect of parameter size on performance metrics. Scatter plots showing the relationship between parameter size (in billions) and performance metrics across 3 prompting strategies (no-role, zero-shot, and few-shot). Solid lines represent fixed-effects predictions from linear mixed-effects models (LMMs) with random intercepts for symptoms, including the main effect of parameter size, prompting strategy, and their interaction. Shaded areas indicate 95% CIs for the predictions: (A) <italic>F</italic><sub>1</sub>-score, (B) balanced accuracy, (C) accuracy, (D) precision, (E) recall, and (F) specificity. Asterisks denote the significance of the parameter size main effect (LMM: *<italic>P</italic>&#x003C;.05, **<italic>P</italic>&#x003C;.01, ***<italic>P</italic>&#x003C;.001).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e98580_fig04.png"/></fig></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This study systematically evaluated 12 open-source LLMs across 4 model families (Qwen3, DeepSeek-R1, Gemma3, and Llama) for extracting intestinal symptoms from EHRs. Qwen3-8b with no-role prompting achieved the highest macroaveraged <italic>F</italic><sub>1</sub>-score of 0.89 with a moderate inference time of 3.64 seconds per record. Although no universal <italic>F</italic><sub>1</sub>-score threshold exists for clinical deployment, prior studies have considered scores between 0.80 and 0.84 acceptable for similar tasks [<xref ref-type="bibr" rid="ref21">21</xref>]. Our observed <italic>F</italic><sub>1</sub>-score of 0.89, combined with a recall of 0.91 and precision of 0.88, suggests potential for population-level surveillance. Prospective validation in real-time settings and use-case&#x2013;specific calibration are needed before full deployment.</p></sec><sec id="s4-2"><title>Prompting Strategy Effects Varied Across Tasks and Models</title><p>The findings revealed that the effect of prompting strategy varied across models, with no single strategy consistently outperforming the others. Although some pairwise comparisons reached statistical significance, the absolute differences in metrics were small, and the direction of improvement was not consistent across models.</p><p>The effect of prompting strategies was most pronounced for the small models in our evaluation, such as Qwen3-1.7b and DeepSeek-R1-1.5b, where zero-shot and few-shot prompting produced more noticeable gains than for larger models. This observation aligns with a broader trend in the literature: prompting techniques tend to yield larger relative improvements in smaller language models, where the baseline zero-shot performance is lower, and the parameter space is more limited [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. For larger models, the absolute differences among prompting strategies were smaller, suggesting that their stronger pretrained representations may already capture much of the task-relevant information without additional in-context guidance [<xref ref-type="bibr" rid="ref24">24</xref>]. From a practical perspective, this finding has important implications for resource-constrained deployments, and our results suggest that prompt design can partially compensate for their smaller parameter capacity. At the same time, the modest effect of prompting strategies across most models indicates that deploying LLMs in such settings may not require extensive prompt optimization. Concise, straightforward instructions may be adequate.</p><p>These results contrast with previous studies demonstrating substantial gains from few-shot prompting in clinical settings [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>], but align with recent work showing that the effectiveness of prompting strategies is highly model-dependent. For instance, Sivarajkumar et al [<xref ref-type="bibr" rid="ref27">27</xref>] found that while heuristic and chain-of-thought prompts excelled in certain tasks, no single prompting strategy universally outperformed the others across the 5 clinical tasks examined. Similarly, a systematic review concluded that applying performance-improvement strategies, including few-shot prompting, may, in some cases, even degrade performance [<xref ref-type="bibr" rid="ref28">28</xref>]. The model-dependent nature of prompting effects may suggest that the benefits of sophisticated prompt engineering may be modest for binary classification tasks such as symptom extraction [<xref ref-type="bibr" rid="ref29">29</xref>]. Alternatively, it may indicate that the few-shot exemplars used were insufficient to capture the nuanced patterns across the target symptoms [<xref ref-type="bibr" rid="ref30">30</xref>].</p></sec><sec id="s4-3"><title>Performance Disparities by Symptom Prevalence</title><p>The analysis revealed that extraction performance varied considerably by symptom type. LLMs pretrained on standard clinical corpora may therefore have weaker representations for symptoms with variable expressions. The lower extraction accuracy for rare symptoms reflects inherent bias in pretraining corpora, which plays a foundational role. LLMs learn statistical regularities from vast text collections that reflect real-world prevalence patterns. Common symptoms such as fever and abdominal pain are overrepresented in medical literature and clinical notes, giving models internal representations for frequent entities [<xref ref-type="bibr" rid="ref31">31</xref>]. Rarer symptoms achieve substantially lower extraction accuracy. As a previous study demonstrated, class imbalance and missingness of signs and symptoms systematically degrade model performance [<xref ref-type="bibr" rid="ref32">32</xref>]. Xi et al [<xref ref-type="bibr" rid="ref26">26</xref>] observed that symptom extraction remains the most challenging entity type in rare disease named entity recognition because symptom labels are context-dependent and often overlap with objective findings, exacerbating the impact of data scarcity.</p><p>Model architecture and pretraining data composition matter more than parameter count for rare symptom recognition. McMurry et al [<xref ref-type="bibr" rid="ref33">33</xref>] further demonstrated that while LLMs significantly outperformed <italic>International Classification of Diseases</italic>, Tenth Revision (<italic>ICD-10</italic>)&#x2013;based methods for respiratory symptom identification, performance variability across symptoms persisted, with rarer symptoms remaining more difficult to extract. Our finding that Qwen3-1.7b outperformed several larger models suggests that model architecture is important for rare symptom recognition. For surveillance systems targeting rare or emerging symptoms, additional strategies such as targeted data augmentation or retrieval-augmented generation may be needed to complement prompt-based approaches. Our prompt-based workflow is model-agnostic and requires no code modification when switching between LLMs, offering a readily deployable solution that can leverage future improved models.</p></sec><sec id="s4-4"><title>Parameter-Size Improvements in Balanced Accuracy</title><p>LLMs revealed that models with larger parameter sizes primarily enhance sensitivity, the ability to correctly identify true symptoms. Previous studies have demonstrated that increasing model size improves generalization and reasoning capabilities, which may translate into better identification of symptoms in clinical text [<xref ref-type="bibr" rid="ref33">33</xref>]. The significant effect on recall is particularly important in clinical contexts where missing a symptom may have more serious consequences than false alarms [<xref ref-type="bibr" rid="ref34">34</xref>].</p><p>The time-to-completion analysis revealed a substantial trade-off between performance and inference efficiency, with larger models incurring markedly longer inference times. The 14b models required an average of 11.6 seconds per chief complaint, compared to 0.5 seconds for 1b models. The findings align with a previous study, which demonstrated that medium-sized Llama models (7b-8b) achieve competitive performance while running up to 28 times faster than 70b counterparts [<xref ref-type="bibr" rid="ref31">31</xref>]. This efficiency gap underscores the practical constraints of deploying deep learning models in resource-limited clinical settings [<xref ref-type="bibr" rid="ref35">35</xref>] and highlights the potential of medium-sized models to balance accuracy with throughput.</p><p>However, inference time did not always increase with model size. For short-context classification tasks, inference latency is not solely determined by parameter count. Hybrid attention mechanisms, quantization, and runtime optimizations can enable larger models to run faster than smaller, less efficient ones. These observations suggest that model selection for deployment in resource-constrained settings should consider not only parameter size but also architectural efficiency and quantization compatibility.</p></sec><sec id="s4-5"><title>Model Architecture Matters More Than Parameter Size Alone</title><p>Our findings reveal that model architecture and training paradigms exert a stronger influence on symptom recognition performance than parameter size alone. While increasing parameter count generally improves performance, the gains exhibit diminishing returns, with the difference between 8b and 14b models failing to reach statistical significance in our analysis. This pattern aligns with recent observations in medical AI evaluation, where architectural innovations can yield substantial performance advantages that are not solely attributable to model scale [<xref ref-type="bibr" rid="ref36">36</xref>].</p><p>The poor performance of Llama2-Chinese further underscores this point. Llama2-Chinese was only posttrained on general Chinese dialogue data, not on medical corpora. Effective adaptation to the Chinese clinical domain reportedly requires continued pretraining. This pattern also extended to Llama3.1-8b, which achieved macro <italic>F</italic><sub>1</sub>-scores that were modestly better than those of Llama2-Chinese but still below the performance of Qwen3 models. These findings suggest that Llama-series models may be less suitable for the Chinese intestinal symptom extraction task.</p><p>We further observed that model architecture can influence performance as profoundly as parameter count. In our symptom extraction task, medium-sized Qwen3 models (1.7b and 8b) outperformed larger models from Llama2-Chinese. Similar patterns have emerged in other clinical information extraction studies, which reported that when extracting social determinants of health from EHRs, open-source models such as OpenChat-3.5 (approximately 7b parameters) consistently outperformed the baseline, while comparably sized Llama-2 models exhibited inferior performance [<xref ref-type="bibr" rid="ref34">34</xref>]. Collectively, these observations indicate that parameter size alone does not determine accuracy; architectural design, domain relevance of pretraining data, and instruction-tuning strategies are also critical [<xref ref-type="bibr" rid="ref37">37</xref>]. For unstructured tasks such as clinical notes, well-engineered medium-sized models can achieve deployable performance in resource-constrained settings, whereas simply increasing model size may yield diminishing returns [<xref ref-type="bibr" rid="ref38">38</xref>-<xref ref-type="bibr" rid="ref40">40</xref>].</p></sec><sec id="s4-6"><title>Clinical Implications</title><p>Our local deployment strategy ensures that patient data remain within the institutions, countering the privacy risks associated with commercial health data brokerage [<xref ref-type="bibr" rid="ref41">41</xref>]. By avoiding external data transmission, this approach upholds patient confidentiality while maintaining competitive model performance, offering a sustainable pathway for AI integration that aligns with emerging regulatory frameworks prioritizing data transparency and patient autonomy. The findings suggest that for straightforward symptom extraction tasks, moderate-sized models (3-8b parameters) may offer the optimal balance between accuracy and efficiency.</p><p>Deployment decisions should balance extraction performance and inference time according to the specific clinical workflow, whether processing is done in real time or in batches, and the acceptable latency for the intended use case. Our inference measurements were performed with a batch size of 1 to reflect single-record latency. In practice, when processing high volumes of data, batch processing can substantially reduce the average time per record by amortizing model initialization and GPU kernel launch overhead. For retrospective symptom extraction in an outpatient clinic, the Qwen3-8b model may be appropriate given its higher <italic>F</italic><sub>1</sub>-score. For real-time applications such as emergency department triage or outbreak alert systems, lower latency may be necessary. Alternatively, the faster Qwen3-1.7b model offers a favorable trade-off between speed and accuracy.</p><p>The limited and model-dependent nature of prompting effects also has practical implications. For symptom classification tasks, clinicians and researchers may not need to invest substantial effort in prompt optimization, as simple, clear instructions appear sufficient to achieve near-optimal performance in some configurations. This observation aligns with recent findings in clinical text classification [<xref ref-type="bibr" rid="ref42">42</xref>].</p></sec><sec id="s4-7"><title>Limitations</title><p>Several limitations warrant consideration. First, the absence of external validation across different clinical settings or languages constrains the conclusions about model generalizability. Our results may not extend to non-Chinese EHRs, to other health care institutions with different documentation practices, or to symptom categories other than intestinal symptoms (eg, respiratory or neurological symptoms) [<xref ref-type="bibr" rid="ref34">34</xref>]. Second, we only used the chief complaint field, which is typically short. Symptom extraction from longer clinical notes, such as the history of present illness or physical examination, may yield different performance and should be investigated in future work. In addition, precision and <italic>F</italic><sub>1</sub>-score capture false positives but do not explicitly quantify hallucination rates (ie, the proportion of extracted symptoms not mentioned in the input text); future work should incorporate dedicated hallucination metrics to assess clinical trustworthiness. Furthermore, the lack of domain-specific fine-tuning means our results represent out-of-the-box performance [<xref ref-type="bibr" rid="ref43">43</xref>]. We did not perform fine-tuning because our study focused on evaluating the capability of open-source general LLMs for symptom extraction under realistic conditions where rapid deployment without task-specific training is required. This workflow is model-agnostic and can be reapplied to future LLMs without code modification or additional annotation. Finally, the time analysis was conducted on a specific hardware configuration; inference times on other hardware (especially newer GPUs) and across different deployment environments may differ substantially and should be interpreted with caution.</p><p>Future work should extend the evaluation to more diverse symptom types and clinical settings, and leverage increasingly capable general-purpose LLMs with richer medical pretraining, which can be integrated into our workflow. Our pipeline is positioned to integrate such newer open-source general-purpose LLMs without additional fine-tuning, offering a sustainable path for keeping pace with rapid LLM advances. Retrieval-augmented generation approaches may further reduce hallucinations and improve grounding.</p></sec><sec id="s4-8"><title>Conclusions</title><p>In this systematic evaluation of open-source LLMs for intestinal symptom extraction, we found that model architecture and parameter size significantly influenced accuracy, with Qwen models offering a favorable trade-off between performance and inference efficiency. Few-shot prompting provided modest gains for rare symptoms but did not consistently outperform simpler instructions across all models or metrics. These findings suggest that for binary symptom classification, moderate-sized models (3-8b) with clear baseline prompts may be sufficient for many resource-constrained clinical deployments.</p></sec></sec></body><back><ack><p>We thank the staff members of the Information Department at the Central Hospital of Wuhan for their assistance with data verification for this study. Generative AI tool (DeepSeek-V3) was used solely for language polishing and grammatical refinement during the preparation of the manuscript. No AI tools were used for data analysis, interpretation, or the generation of scientific content.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the National Key Research and Development Program of China (grant 2022YFC2305103).</p></sec><sec><title>Data Availability</title><p>The data that support the findings of this study are available from the Health Information Center of Wuhan, but restrictions apply to the availability of these data, which were used under license for this study and so are not publicly available. Due to the sensitive nature of patient data and privacy protection requirements, the electronic health records supporting this study are not publicly available.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: XZ, QW, BL, XS, SW</p><p>Data curation: XZ, QW, BL, XS</p><p>Design: XZ, QW, BL, XS, SW</p><p>Formal analysis: XZ, QW</p><p>Funding acquisition: SW</p><p>Methodology: XZ, QW, BL, XS, SW</p><p>Project administration: SW</p><p>Software: XZ, QW, BL, XS</p><p>Supervision: SW</p><p>Writing-original draft: XZ, QW</p><p>Writing-review and editing: XZ, SW</p><p>All authors read and approved the final manuscript.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">EHR</term><def><p>electronic health record</p></def></def-item><def-item><term id="abb2">FDR</term><def><p>false discovery rate</p></def></def-item><def-item><term id="abb3">GPU</term><def><p>graphical processing unit</p></def></def-item><def-item><term id="abb4"><italic>ICD-10</italic></term><def><p><italic>International Classification of Diseases</italic>, Tenth Revision</p></def></def-item><def-item><term id="abb5">IID</term><def><p>intestinal infectious disease</p></def></def-item><def-item><term id="abb6">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb7">LMM</term><def><p>linear mixed model</p></def></def-item><def-item><term id="abb8">STARD-AI</term><def><p>Standards for Reporting Diagnostic Accuracy Studies&#x2013;AI</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><collab>GBD 2023 Diarrhoeal Disease and Enteric Infectious Diseases Collaborators</collab></person-group><article-title>Global burden of enteric infectious diseases, diarrhoeal diseases, and corresponding aetiologies, 1990-2023: a systematic analysis for the Global Burden of Disease Study 2023</article-title><source>Lancet Infect Dis</source><year>2026</year><month>06</month><day>2</day><fpage>S1473-3099(26)00194-5</fpage><pub-id pub-id-type="doi">10.1016/S1473-3099(26)00194-5</pub-id><pub-id pub-id-type="medline">42229499</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sim</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Horan</surname><given-names>MR</given-names> </name><etal/></person-group><article-title>Natural language processing with machine learning methods to analyze unstructured patient-reported outcomes derived from electronic health records: a systematic review</article-title><source>Artif Intell Med</source><year>2023</year><month>12</month><volume>146</volume><fpage>102701</fpage><pub-id pub-id-type="doi">10.1016/j.artmed.2023.102701</pub-id><pub-id pub-id-type="medline">38042599</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Seinen</surname><given-names>TM</given-names> </name><name name-style="western"><surname>Kors</surname><given-names>JA</given-names> </name><name name-style="western"><surname>van Mulligen</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Rijnbeek</surname><given-names>PR</given-names> </name></person-group><article-title>Using structured codes and free-text notes to measure information complementarity in electronic health records: feasibility and validation study</article-title><source>J Med Internet Res</source><year>2025</year><month>02</month><day>13</day><volume>27</volume><fpage>e66910</fpage><pub-id pub-id-type="doi">10.2196/66910</pub-id><pub-id pub-id-type="medline">39946687</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Struyf</surname><given-names>T</given-names> </name><name name-style="western"><surname>Deeks</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Dinnes</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Signs and symptoms to determine if a patient presenting in primary care or hospital outpatient settings has COVID-19</article-title><source>Cochrane Database Syst Rev</source><year>2022</year><month>05</month><day>20</day><volume>5</volume><issue>5</issue><fpage>CD013665</fpage><pub-id pub-id-type="doi">10.1002/14651858.CD013665.pub3</pub-id><pub-id pub-id-type="medline">35593186</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Price</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Stapley</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Shephard</surname><given-names>E</given-names> </name><name name-style="western"><surname>Barraclough</surname><given-names>K</given-names> </name><name name-style="western"><surname>Hamilton</surname><given-names>WT</given-names> </name></person-group><article-title>Is omission of free text records a possible source of data loss and bias in Clinical Practice Research Datalink studies? A case-control study</article-title><source>BMJ Open</source><year>2016</year><month>05</month><day>13</day><volume>6</volume><issue>5</issue><fpage>e011664</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2016-011664</pub-id><pub-id pub-id-type="medline">27178981</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thakkar</surname><given-names>V</given-names> </name><name name-style="western"><surname>Silverman</surname><given-names>GM</given-names> </name><name name-style="western"><surname>Kc</surname><given-names>A</given-names> </name><etal/></person-group><article-title>A comparative analysis of large language models versus traditional information extraction methods for real-world evidence of patient symptomatology in acute and post-acute sequelae of SARS-CoV-2</article-title><source>PLoS One</source><year>2025</year><volume>20</volume><issue>5</issue><fpage>e0323535</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0323535</pub-id><pub-id pub-id-type="medline">40373001</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bejan</surname><given-names>CA</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>M</given-names> </name><name name-style="western"><surname>Venkateswaran</surname><given-names>S</given-names> </name><etal/></person-group><article-title>irAE-GPT: leveraging large language models to identify immune-related adverse events in electronic health records and clinical trial datasets</article-title><source>EBioMedicine</source><year>2026</year><month>05</month><volume>127</volume><fpage>106227</fpage><pub-id pub-id-type="doi">10.1016/j.ebiom.2026.106227</pub-id><pub-id pub-id-type="medline">41951517</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Qin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Feng</surname><given-names>L</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Han</surname><given-names>L</given-names> </name></person-group><article-title>ZeroTuneBio NER: a three-stage framework for zero-shot and zero-tuning biomedical entity extraction using large language models and prompt engineering</article-title><source>Comput Methods Programs Biomed</source><year>2025</year><month>12</month><volume>272</volume><fpage>109070</fpage><pub-id pub-id-type="doi">10.1016/j.cmpb.2025.109070</pub-id><pub-id pub-id-type="medline">40945000</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Knevel</surname><given-names>R</given-names> </name><name name-style="western"><surname>Liao</surname><given-names>KP</given-names> </name></person-group><article-title>From real-world electronic health record data to real-world results using artificial intelligence</article-title><source>Ann Rheum Dis</source><year>2023</year><month>03</month><volume>82</volume><issue>3</issue><fpage>306</fpage><lpage>311</lpage><pub-id pub-id-type="doi">10.1136/ard-2022-222626</pub-id><pub-id pub-id-type="medline">36150748</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Park</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wee</surname><given-names>CW</given-names> </name><name name-style="western"><surname>Choi</surname><given-names>SH</given-names> </name><etal/></person-group><article-title>Improving mortality prediction after radiotherapy with large language model structuring of large-scale unstructured electronic health records</article-title><source>Radiother Oncol</source><year>2025</year><month>10</month><volume>211</volume><fpage>111052</fpage><pub-id pub-id-type="doi">10.1016/j.radonc.2025.111052</pub-id><pub-id pub-id-type="medline">40692078</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhou</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>C</given-names> </name><etal/></person-group><article-title>A suite of large language models for public health infoveillance</article-title><source>NPJ Digit Med</source><year>2026</year><month>02</month><day>23</day><volume>9</volume><issue>1</issue><fpage>270</fpage><pub-id pub-id-type="doi">10.1038/s41746-026-02435-6</pub-id><pub-id pub-id-type="medline">41731011</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Woo</surname><given-names>EG</given-names> </name><name name-style="western"><surname>Burkhart</surname><given-names>MC</given-names> </name><name name-style="western"><surname>Alsentzer</surname><given-names>E</given-names> </name><name name-style="western"><surname>Beaulieu-Jones</surname><given-names>BK</given-names> </name></person-group><article-title>Synthetic data distillation enables the extraction of clinical information at scale</article-title><source>NPJ Digit Med</source><year>2025</year><month>05</month><day>10</day><volume>8</volume><issue>1</issue><fpage>267</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01681-4</pub-id><pub-id pub-id-type="medline">40348936</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Xiong</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zou</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Reasoning-driven large language models in medicine: opportunities, challenges, and the road ahead</article-title><source>Lancet Digit Health</source><year>2026</year><month>01</month><volume>8</volume><issue>1</issue><fpage>100931</fpage><pub-id pub-id-type="doi">10.1016/j.landig.2025.100931</pub-id><pub-id pub-id-type="medline">41620322</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhong</surname><given-names>W</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Performance of ChatGPT-4o and four open-source large language models in generating diagnoses based on China&#x2019;s Rare Disease Catalog: comparative study</article-title><source>J Med Internet Res</source><year>2025</year><month>06</month><day>18</day><volume>27</volume><fpage>e69929</fpage><pub-id pub-id-type="doi">10.2196/69929</pub-id><pub-id pub-id-type="medline">40532199</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hart</surname><given-names>SN</given-names> </name><name name-style="western"><surname>Bergamaschi</surname><given-names>TS</given-names> </name></person-group><article-title>Agent-based large language model system for extracting structured data from breast cancer synoptic reports: a dual-validation study</article-title><source>JAMIA Open</source><year>2026</year><month>02</month><volume>9</volume><issue>1</issue><fpage>ooag016</fpage><pub-id pub-id-type="doi">10.1093/jamiaopen/ooag016</pub-id><pub-id pub-id-type="medline">41756715</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Touvron</surname><given-names>H</given-names> </name><name name-style="western"><surname>Lavril</surname><given-names>T</given-names> </name><name name-style="western"><surname>Izacard</surname><given-names>G</given-names> </name><name name-style="western"><surname>Martinet</surname><given-names>X</given-names> </name><etal/></person-group><article-title>LLaMA: open and efficient foundation language models</article-title><source>arXiv</source><comment>Preprint posted online on  Feb 27, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2302.13971</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jiang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Zeng</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>C</given-names> </name></person-group><article-title>Promoting responsible DeepSeek deployment in health care: scoping review comparing grey and white literature</article-title><source>J Med Internet Res</source><year>2025</year><month>12</month><day>5</day><volume>27</volume><fpage>e80770</fpage><pub-id pub-id-type="doi">10.2196/80770</pub-id><pub-id pub-id-type="medline">41348941</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>T</given-names> </name><name name-style="western"><surname>Xiao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Bao</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Hao</surname><given-names>J</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>J</given-names> </name></person-group><article-title>The rise and potential opportunities of large language model agents in bioinformatics and biomedicine</article-title><source>Brief Bioinform</source><year>2025</year><month>11</month><day>1</day><volume>26</volume><issue>6</issue><fpage>bbaf601</fpage><pub-id pub-id-type="doi">10.1093/bib/bbaf601</pub-id><pub-id pub-id-type="medline">41214870</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Omar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sorin</surname><given-names>V</given-names> </name><name name-style="western"><surname>Collins</surname><given-names>JD</given-names> </name><etal/></person-group><article-title>Multi-model assurance analysis showing large language models are highly vulnerable to adversarial hallucination attacks during clinical decision support</article-title><source>Commun Med (Lond)</source><year>2025</year><month>08</month><day>2</day><volume>5</volume><issue>1</issue><fpage>330</fpage><pub-id pub-id-type="doi">10.1038/s43856-025-01021-3</pub-id><pub-id pub-id-type="medline">40753316</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Windisch</surname><given-names>P</given-names> </name><name name-style="western"><surname>Dennst&#x00E4;dt</surname><given-names>F</given-names> </name><name name-style="western"><surname>Koechli</surname><given-names>C</given-names> </name><etal/></person-group><article-title>The impact of temperature on extracting information from clinical trial publications using large language models</article-title><source>Cureus</source><year>2024</year><month>12</month><volume>16</volume><issue>12</issue><fpage>e75748</fpage><pub-id pub-id-type="doi">10.7759/cureus.75748</pub-id><pub-id pub-id-type="medline">39811231</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McMurry</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Zipursky</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Geva</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Moving biosurveillance beyond coded data using AI for symptom detection from physician notes: retrospective cohort study</article-title><source>J Med Internet Res</source><year>2024</year><month>04</month><day>4</day><volume>26</volume><fpage>e53367</fpage><pub-id pub-id-type="doi">10.2196/53367</pub-id><pub-id pub-id-type="medline">38573752</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhao</surname><given-names>F</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>M</given-names> </name><name name-style="western"><surname>Luo</surname><given-names>C</given-names> </name></person-group><article-title>A comparative empirical study of prompting strategies for code generation with large language models</article-title><source>J Adv Comput Syst</source><year>2025</year><volume>5</volume><issue>12</issue><fpage>26</fpage><lpage>37</lpage><pub-id pub-id-type="doi">10.69987/JACS.2025.51203</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jelodar</surname><given-names>H</given-names> </name><name name-style="western"><surname>Meymani</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hamedi</surname><given-names>P</given-names> </name><etal/></person-group><article-title>NLD-LLM: a systematic framework for evaluating small language transformer models on natural language description</article-title><source>2025 Int Conf Mach Learn Appl (ICMLA)</source><year>2025</year><fpage>1494</fpage><lpage>1500</lpage><pub-id pub-id-type="doi">10.1109/ICMLA66185.2025.00227</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Palmetshofer</surname><given-names>M</given-names> </name><name name-style="western"><surname>Schedl</surname><given-names>DC</given-names> </name><name name-style="western"><surname>St&#x00F6;ckl</surname><given-names>A</given-names> </name></person-group><article-title>Optimizing app review classification with large language models: a comparative study of prompting techniques</article-title><source>2024 4th Int Conf Electr Comput Commun Mechatronics Eng (ICECCME)</source><year>2024</year><fpage>1</fpage><lpage>6</lpage><pub-id pub-id-type="doi">10.1109/ICECCME62383.2024.10796376</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shao</surname><given-names>C</given-names> </name><name name-style="western"><surname>Snyder</surname><given-names>D</given-names> </name><name name-style="western"><surname>Li</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Scalable medication extraction and discontinuation identification from electronic health records using large language models</article-title><source>J Clin Epidemiol</source><year>2026</year><month>01</month><volume>189</volume><fpage>112049</fpage><pub-id pub-id-type="doi">10.1016/j.jclinepi.2025.112049</pub-id><pub-id pub-id-type="medline">41232578</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xi</surname><given-names>NM</given-names> </name><name name-style="western"><surname>Deng</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name></person-group><article-title>Leveraging large language models for rare disease named entity recognition</article-title><source>PLoS Digit Health</source><year>2026</year><month>02</month><volume>5</volume><issue>2</issue><fpage>e0001242</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0001242</pub-id><pub-id pub-id-type="medline">41678536</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sivarajkumar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kelley</surname><given-names>M</given-names> </name><name name-style="western"><surname>Samolyk-Mazzanti</surname><given-names>A</given-names> </name><name name-style="western"><surname>Visweswaran</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name></person-group><article-title>An empirical evaluation of prompting strategies for large language models in zero-shot clinical natural language processing: algorithm development and validation study</article-title><source>JMIR Med Inform</source><year>2024</year><month>04</month><day>8</day><volume>12</volume><fpage>e55318</fpage><pub-id pub-id-type="doi">10.2196/55318</pub-id><pub-id pub-id-type="medline">38587879</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Du</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Performance and improvement strategies for adapting generative large language models for electronic health record applications: a systematic review</article-title><source>Int J Med Inform</source><year>2026</year><month>01</month><volume>205</volume><fpage>106091</fpage><pub-id pub-id-type="doi">10.1016/j.ijmedinf.2025.106091</pub-id><pub-id pub-id-type="medline">40885071</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zheng</surname><given-names>C</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Verification is all you need: prompting large language models for zero-shot clinical coding</article-title><source>IEEE J Biomed Health Inform</source><year>2025</year><month>11</month><volume>29</volume><issue>11</issue><fpage>8536</fpage><lpage>8549</lpage><pub-id pub-id-type="doi">10.1109/JBHI.2025.3593028</pub-id><pub-id pub-id-type="medline">40720269</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Owens</surname><given-names>D</given-names> </name><name name-style="western"><surname>Nguyen</surname><given-names>DQ</given-names> </name><name name-style="western"><surname>Dohopolski</surname><given-names>M</given-names> </name><name name-style="western"><surname>Rousseau</surname><given-names>JF</given-names> </name><name name-style="western"><surname>Peterson</surname><given-names>ED</given-names> </name><name name-style="western"><surname>Navar</surname><given-names>AM</given-names> </name></person-group><article-title>Accuracy of large language models to identify stroke subtypes within unstructured electronic health record data</article-title><source>Stroke</source><year>2025</year><month>10</month><volume>56</volume><issue>10</issue><fpage>2966</fpage><lpage>2975</lpage><pub-id pub-id-type="doi">10.1161/STROKEAHA.125.051993</pub-id><pub-id pub-id-type="medline">40709446</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zuo</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Information extraction from clinical notes: are we ready to switch to large language models?</article-title><source>J Am Med Inform Assoc</source><year>2026</year><month>03</month><day>1</day><volume>33</volume><issue>3</issue><fpage>553</fpage><lpage>562</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocaf213</pub-id><pub-id pub-id-type="medline">41533750</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Spiero</surname><given-names>I</given-names> </name><name name-style="western"><surname>Rijk</surname><given-names>MH</given-names> </name><name name-style="western"><surname>Scheeres</surname><given-names>MA</given-names> </name><etal/></person-group><article-title>Comparison of local large language models for extraction of signs and symptoms data from electronic health records</article-title><source>PLoS One</source><year>2026</year><volume>21</volume><issue>6</issue><fpage>e0350625</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0350625</pub-id><pub-id pub-id-type="medline">42268852</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McMurry</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Phelan</surname><given-names>D</given-names> </name><name name-style="western"><surname>Dixon</surname><given-names>BE</given-names> </name><etal/></person-group><article-title>Large language model symptom identification from clinical text: multicenter study</article-title><source>J Med Internet Res</source><year>2025</year><month>07</month><day>31</day><volume>27</volume><fpage>e72984</fpage><pub-id pub-id-type="doi">10.2196/72984</pub-id><pub-id pub-id-type="medline">40743494</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rhee</surname><given-names>JY</given-names> </name><name name-style="western"><surname>Tentor</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Sounack</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Scalable tracking of symptoms in the electronic health record using large language models in patients with central nervous system cancers undergoing therapy</article-title><source>Neuro Oncol</source><year>2026</year><month>01</month><day>1</day><volume>28</volume><issue>1</issue><fpage>206</fpage><lpage>217</lpage><pub-id pub-id-type="doi">10.1093/neuonc/noaf223</pub-id><pub-id pub-id-type="medline">40973180</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>RY</given-names> </name><name name-style="western"><surname>Kross</surname><given-names>EK</given-names> </name><name name-style="western"><surname>Torrence</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Assessment of natural language processing of electronic health records to measure goals-of-care discussions as a clinical trial outcome</article-title><source>JAMA Netw Open</source><year>2023</year><month>03</month><day>1</day><volume>6</volume><issue>3</issue><fpage>e231204</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2023.1204</pub-id><pub-id pub-id-type="medline">36862411</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bedi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Cui</surname><given-names>H</given-names> </name><name name-style="western"><surname>Fuentes</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Holistic evaluation of large language models for medical tasks with MedHELM</article-title><source>Nat Med</source><year>2026</year><month>03</month><volume>32</volume><issue>3</issue><fpage>943</fpage><lpage>951</lpage><pub-id pub-id-type="doi">10.1038/s41591-025-04151-2</pub-id><pub-id pub-id-type="medline">41559415</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Li</surname><given-names>ZZ</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>ML</given-names> </name><etal/></person-group><article-title>From system 1 to system 2: a survey of reasoning large language models</article-title><source>IEEE Trans Pattern Anal Mach Intell</source><year>2026</year><month>03</month><volume>48</volume><issue>3</issue><fpage>3335</fpage><lpage>3354</lpage><pub-id pub-id-type="doi">10.1109/TPAMI.2025.3637037</pub-id><pub-id pub-id-type="medline">41289126</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wiest</surname><given-names>IC</given-names> </name><name name-style="western"><surname>Wolf</surname><given-names>F</given-names> </name><name name-style="western"><surname>Le&#x00DF;mann</surname><given-names>ME</given-names> </name><etal/></person-group><article-title>A software pipeline for medical information extraction with large language models, open source and suitable for oncology</article-title><source>NPJ Precis Oncol</source><year>2025</year><month>09</month><day>17</day><volume>9</volume><issue>1</issue><fpage>313</fpage><pub-id pub-id-type="doi">10.1038/s41698-025-01103-4</pub-id><pub-id pub-id-type="medline">40962856</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>RJ</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Tsai</surname><given-names>LW</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Shen Hsiao</surname><given-names>ST</given-names> </name><name name-style="western"><surname>Lo</surname><given-names>YS</given-names> </name></person-group><article-title>Integrating a large language model to streamline nursing handover documentation across multiple hospitals in Taiwan: development and implementation study</article-title><source>J Med Internet Res</source><year>2026</year><month>03</month><day>12</day><volume>28</volume><fpage>e81604</fpage><pub-id pub-id-type="doi">10.2196/81604</pub-id><pub-id pub-id-type="medline">41819121</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tordjman</surname><given-names>M</given-names> </name><name name-style="western"><surname>Yuce</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ammar</surname><given-names>A</given-names> </name><etal/></person-group><article-title>The rise of deepfake medical imaging: radiologists&#x2019; diagnostic accuracy in detecting ChatGPT-generated radiographs</article-title><source>Radiology</source><year>2026</year><month>03</month><volume>318</volume><issue>3</issue><fpage>e252094</fpage><pub-id pub-id-type="doi">10.1148/radiol.252094</pub-id><pub-id pub-id-type="medline">41874300</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Crowson</surname><given-names>MG</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>JZH</given-names> </name><name name-style="western"><surname>Dunn</surname><given-names>J</given-names> </name><etal/></person-group><article-title>The need to develop health data transaction disclosure requirements to balance transparency, privacy, and progressive use</article-title><source>Lancet Digit Health</source><year>2026</year><month>02</month><volume>8</volume><issue>2</issue><fpage>100947</fpage><pub-id pub-id-type="doi">10.1016/j.landig.2025.100947</pub-id><pub-id pub-id-type="medline">41792017</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Naliyatthaliyazchayil</surname><given-names>P</given-names> </name><name name-style="western"><surname>Muthyala</surname><given-names>R</given-names> </name><name name-style="western"><surname>Gichoya</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Purkayastha</surname><given-names>S</given-names> </name></person-group><article-title>Evaluating the reasoning capabilities of large language models for medical coding and hospital readmission risk stratification: zero-shot prompting approach</article-title><source>J Med Internet Res</source><year>2025</year><month>07</month><day>30</day><volume>27</volume><fpage>e74142</fpage><pub-id pub-id-type="doi">10.2196/74142</pub-id><pub-id pub-id-type="medline">40737604</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhou</surname><given-names>W</given-names> </name><name name-style="western"><surname>Yetisgen</surname><given-names>M</given-names> </name><name name-style="western"><surname>Afshar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Savova</surname><given-names>G</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>TA</given-names> </name></person-group><article-title>Improving model transferability for clinical note section classification models using continued pretraining</article-title><source>J Am Med Inform Assoc</source><year>2023</year><month>12</month><day>22</day><volume>31</volume><issue>1</issue><fpage>89</fpage><lpage>97</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocad190</pub-id><pub-id pub-id-type="medline">37725927</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Symptoms of intestinal infectious diseases from clinical guidelines.</p><media xlink:href="jmir_v28i1e98580_app1.docx" xlink:title="DOCX File, 44 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Refinement process for extracting target symptoms in a large language model.</p><media xlink:href="jmir_v28i1e98580_app2.docx" xlink:title="DOCX File, 26 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Identification of target symptoms for intestinal infectious diseases.</p><media xlink:href="jmir_v28i1e98580_app3.docx" xlink:title="DOCX File, 358 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Statistics for all models.</p><media xlink:href="jmir_v28i1e98580_app4.docx" xlink:title="DOCX File, 339 KB"/></supplementary-material><supplementary-material id="app5"><label>Multimedia Appendix 5</label><p>Confusion matrices for all models across the 6 target symptoms.</p><media xlink:href="jmir_v28i1e98580_app5.docx" xlink:title="DOCX File, 993 KB"/></supplementary-material><supplementary-material id="app6"><label>Checklist 1</label><p>STARD-AI checklist.</p><media xlink:href="jmir_v28i1e98580_app6.pdf" xlink:title="PDF File, 75 KB"/></supplementary-material></app-group></back></article>