<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e96304</article-id><article-id pub-id-type="doi">10.2196/96304</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>A Multiagent Large Language Model Framework for Emergency Treatment Recommendation in Acute Ischemic Stroke: Development and Validation Study</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Yan</surname><given-names>Bicong</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Zhang</surname><given-names>Ruipeng</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chen</surname><given-names>Li</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Song</surname><given-names>Xinyu</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Cao</surname><given-names>Zhongzheng</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Li</surname><given-names>Yuehua</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Diagnostic and Interventional Radiology, Shanghai Sixth People's Hospital</institution><addr-line>No.600 Yishan Road, Xuhui District</addr-line><addr-line>Shanghai</addr-line><country>China</country></aff><aff id="aff2"><institution>Faculty of Medical Imaging Technology, College of Health Science and Technology, Shanghai Jiao Tong University</institution><addr-line>Shanghai</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Vundavalli</surname><given-names>Harish</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Vairagade</surname><given-names>Hruday</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Ogunbowale</surname><given-names>Oluwatobilola</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Yuehua Li, MD, PhD, Department of Diagnostic and Interventional Radiology, Shanghai Sixth People's Hospital, No.600 Yishan Road, Xuhui District, Shanghai, 200233, China, 86 18918727305; <email>liyuehua77@sjtu.edu.cn</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>30</day><month>7</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e96304</elocation-id><history><date date-type="received"><day>27</day><month>03</month><year>2026</year></date><date date-type="rev-recd"><day>23</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>23</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Bicong Yan, Ruipeng Zhang, Li Chen, Xinyu Song, Zhongzheng Cao, Yuehua Li. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 30.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e96304"/><abstract><sec><title>Background</title><p>Acute ischemic stroke (AIS) treatment selection requires rapid, guideline-concordant integration of clinical, imaging, and laboratory data, including therapeutic windows, contraindications, stroke severity, and imaging eligibility. This process is complex, expertise-dependent, and vulnerable to safety-critical errors.</p></sec><sec><title>Objective</title><p>This study aimed to develop and validate a structured multiagent large language model (LLM) framework for AIS decision support using real-world cases and to assess its accuracy, safety, auditability, and impact on physician decision-making, particularly among junior physicians and nonspecialists.</p></sec><sec sec-type="methods"><title>Methods</title><p>We developed a multiagent LLM framework that used structured outputs and guideline-based reasoning to generate treatment recommendations (intravenous thrombolysis, endovascular thrombectomy, standard medical therapy, or non-AIS, or nonstroke) and Trial of ORG 10172 in Acute Stroke Treatment (TOAST) classification. The framework was evaluated using multicenter retrospective real-world cases from 2 hospitals collected between January 2018 and March 2025, prospective cases from February to May 2025, and literature-derived challenging cases from PubMed between January 2024 and January 2025. Performance was assessed against clinical reference standards. Safety was assessed using omission and hallucination event rates, instruction adherence, and 5-point clinical safety ratings. In a prospective physician study, physicians with different seniority and specialty backgrounds made AIS treatment and TOAST classification decisions with and without LLM support. Physician-case decision-level outcomes were analyzed using a binomial generalized linear mixed-effects model accounting for physician and case effects.</p></sec><sec sec-type="results"><title>Results</title><p>The final analysis included 1055 group A cases, 721 group B cases, 144 literature-derived group C cases, and 161 prospectively collected group D cases. Across representative Baichuan, Qwen, DeepSeek, and GPT models, the multiagent framework consistently improved treatment recommendation accuracy. Model-level accuracy ranges increased from 0.546&#x2010;0.737 to 0.687&#x2010;0.851 in group A, from 0.587&#x2010;0.698 to 0.671&#x2010;0.813 in group B, and from 0.507&#x2010;0.646 to 0.667&#x2010;0.750 in group C. TOAST classification improved overall, with cohort-level variation. Across evaluated models, the multiagent framework increased the mean clinical safety score from 3.70 to 4.01 and reduced mean hallucination and omission rates from 33.6% to 20.6% and from 38.5% to 24.5%, respectively. In the prospective physician study, LLM support increased treatment decision accuracy from 73.1% to 88.6% (odds ratio 2.86, 95% CI, 2.27&#x2010;3.60; <italic>P</italic>&#x003C;.001). Accuracy gains were largest among junior and nonspecialist physicians, including junior specialists (0.667 to 0.833), junior nonspecialists (0.600 to 0.846), and senior nonspecialists (0.667 to 0.850). TOAST classification performance also improved (odds ratio 3.63, 95% CI 2.85&#x2010;4.64; <italic>P</italic>&#x003C;.001).</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>A structured multiagent framework improved LLM performance with average improvements of 18.9% in AIS treatment recommendation and TOAST classification, while producing more structured, auditable outputs with higher safety ratings. It was associated with higher physician decision accuracy, with larger gains among less-experienced physicians, suggesting the potential to narrow expertise-related decision accuracy gaps. Prospective multicenter studies are needed to assess effects on workflow and clinical outcomes.</p></sec><sec><title>Trial Registration</title><p>Chinese Clinical Trial Registry ChiCTR2400092800; https://www.chictr.org.cn/showprojEN.html?proj=248894</p></sec></abstract><kwd-group><kwd>acute ischemic stroke</kwd><kwd>large language model</kwd><kwd>multiagent system</kwd><kwd>decision support</kwd><kwd>clinical safety</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Stroke remains a leading cause of disability and death worldwide, with the greatest burden borne by low- and middle-income countries [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. While acute ischemic stroke (AIS) management requires rapid and precise decision-making within narrow therapeutic windows, stroke care resources and diagnostic expertise remain profoundly uneven globally. Driven by socioeconomic inequality, geographic barriers, and workforce shortages across both resource-limited and high-income settings, these disparities lead to critical delays and misdiagnoses, worsening patient outcomes and reinforcing inequities in access to timely, specialist-level care [<xref ref-type="bibr" rid="ref3">3</xref>-<xref ref-type="bibr" rid="ref7">7</xref>]. These systemic challenges make equitable access to evidence-based AIS management a pressing global priority. As underscored by the World Stroke Organization Global Declaration on Stroke, which calls for equitable, evidence-based stroke systems worldwide, efforts to expand manpower and training have so far yielded only modest and uneven progress [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>].</p><p>Large language models (LLMs) have shown potential to support clinical information synthesis and decision-making, offering a possible strategy to improve the consistency and accessibility of AIS decision support [<xref ref-type="bibr" rid="ref10">10</xref>]. LLMs have demonstrated strong capabilities in diagnostic reasoning, differential generation, and information synthesis [<xref ref-type="bibr" rid="ref11">11</xref>-<xref ref-type="bibr" rid="ref15">15</xref>]. However, whether LLMs can provide reliable, guideline-concordant, and clinically safe decision support for AIS treatment selection remains insufficiently established. Most existing studies are benchmark- or theory-driven, focusing on tasks such as MedQA (United States Medical Licensing Examination) that do not adequately reflect real-world, guideline-based clinical decision-making [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. This gap underscores the need for systematic case-based evaluation using real-world AIS data and physician decision-making experiments.</p><p>Our framework augments LLMs with structured clinical reasoning to support guideline-concordant therapy selection and more reliable AIS decision-making. This study had 2 objectives: to determine whether a multiagent design improves LLM performance on real-world AIS decision tasks in a multicenter, multisource evaluation, and to assess its effect on physician decision accuracy in a human-AI interaction experiment involving physicians with varying seniority and specialty backgrounds. We further examined whether LLM assistance preferentially benefits less-experienced clinicians and narrows expertise-related decision gaps, thereby supporting more consistent, safe, and accessible stroke care decision support across diverse care settings.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Ethical Considerations</title><p>Approval was obtained from the institutional review board of the Medical Faculty Ethics Committee of Shanghai Sixth People's Hospital affiliated to Shanghai Jiao Tong University School of Medicine (approval 2024-KY-203), and the study was registered in the Chinese Clinical Trial Registry (ChiCTR2400092800) on November 22, 2024. Informed consent was obtained from all participants. This study was conducted in accordance with the Declaration of Helsinki and relevant ethical guidelines and regulations.</p></sec><sec id="s2-2"><title>Study Design</title><p>The overall study design and data sources are summarized in <xref ref-type="fig" rid="figure1">Figure 1</xref>. This schematic highlights the integration of multicenter retrospective and prospective real-world clinical cases with PubMed case reports, the application of the multiagent framework, and the evaluation of both model performance and human-AI interaction.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Study framework. The multiagent framework was benchmarked against standalone large language models (LLMs) and quantitatively validated across retrospective, prospective, multicenter, and PubMed datasets. Safety analyses encompassed harmful content, hallucinations, and numerical robustness, while clinical validation assessed human-AI interactions by comparing physicians&#x2019; decisions across experience levels, specialties, and geographic locations. CoT: chain of thought.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e96304_fig01.png"/></fig></sec><sec id="s2-3"><title>Data Collection</title><p>We retrospectively screened clinical cases from 2 tertiary care centers: 1228 cases from center A (group A, January 2018-January 2025, tertiary grade A hospital) and 938 cases from center B (group B, May 2018-March 2025, tertiary grade B hospital). All cases were deidentified encounters of patients diagnosed with acute cerebrovascular disease. In addition, 327 stroke case reports were retrieved from PubMed between January 2024 and January 2025 (group C), which predominantly represented diagnostically challenging and clinically complex presentations. For prospective validation, 213 patients were consecutively enrolled at center A between February and May 2025 (group D). Detailed inclusion and exclusion criteria for each group are provided in the Supplementary Material Methods section and Figure S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-4"><title>Patient Cases</title><p>To ensure patient privacy, all personally identifiable information was removed. Each case was formatted as a single paragraph containing all or a subset of the following elements: patient age and sex, chief complaint, current symptoms, medical history (including illnesses and medications), relevant family history, physical examination findings, laboratory test results, and imaging reports.</p></sec><sec id="s2-5"><title>Evaluation of Standalone LLMs</title><p>To evaluate LLM performance in generating AIS treatment recommendations and assigning stroke subtypes according to the Trial of ORG 10172 in Acute Stroke Treatment (TOAST) classification, each model was tasked with generating a treatment recommendation and the corresponding TOAST classification conclusion. Seven models were tested: Baichuan-M1-14B [<xref ref-type="bibr" rid="ref18">18</xref>], GPT-OSS-20B [<xref ref-type="bibr" rid="ref19">19</xref>], Qwen2.5-32B [<xref ref-type="bibr" rid="ref20">20</xref>], DeepSeek-R1-Distill-Qwen2.5-32B [<xref ref-type="bibr" rid="ref21">21</xref>], GPT-OSS-120B [<xref ref-type="bibr" rid="ref19">19</xref>], DeepSeek-R1-671B [<xref ref-type="bibr" rid="ref21">21</xref>], and GPT-4o (OpenAI; only in group C dataset) [<xref ref-type="bibr" rid="ref22">22</xref>] (Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Real-world clinical cases from groups A to D were used for this assessment. Outputs were produced in free-text format without predefined options, reflecting the probabilistic nature of clinical reasoning. To ensure consistent and reproducible evaluation, an automated grader agent was used to quantify accuracy across all LLMs (Supplementary Material Methods section in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p><p>All LLMs were evaluated in single-turn interactions using their default parameter configurations, without additional manual tuning. Model inference was performed with the official vLLM framework, providing an optimized environment for efficient large-scale deployment. All models were executed on a cluster of 8 NVIDIA H20-141GB graphics processing units (GPUs) using the official Docker release. For version control, vLLM v0.8.4 was applied to all models except GPT-OSS-20B and GPT-OSS-120B, which were deployed under v0.10.1, thereby ensuring reproducibility and transparency across experiments.</p></sec><sec id="s2-6"><title>The Multiagent Framework</title><sec id="s2-6-1"><title>Overview</title><p>In addition to standalone evaluations, we implemented a multiagent framework to examine whether structured reasoning and constrained outputs could enhance model performance in AIS-specific tasks. The framework integrates three components: (1) a workflow-oriented summary agent that extracts disease-relevant evidence from lengthy clinical narratives, (2) a guideline-concordant reasoning-path chain of thought (Short-CoT) Agent that enforces structured diagnostic steps, and (3) a clinically inspired multiple-choice constraint agent that standardizes outputs within evidence-based decision boundaries.</p></sec><sec id="s2-6-2"><title>Summarization Agent</title><p>To mitigate performance degradation caused by lengthy case tokens and to ensure the extraction of salient details, a workflow-oriented summary agent was implemented. Using DeepSeek-R1 with few-shot prompting, this agent generated structured case summaries for downstream reasoning.</p></sec><sec id="s2-6-3"><title>Short-CoT Agent</title><p>The Short-CoT module was designed as a concise sequence of 4 critical diagnostic steps, mirroring clinical decision trees. It was developed collaboratively with neurologists, interventional radiologists, and emergency physicians by restructuring existing clinical guidelines (refer to the prompt provided in Supplementary Material Methods in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p></sec><sec id="s2-6-4"><title>Multiple-Choice Constraint Agent</title><p>For clinically inspired final decision-making, models augmented by the multiagent framework were evaluated using 4 predefined categories as answer options: <italic>thrombolysis, mechanical thrombectomy, standard medical therapy, and non-AIS or nonstroke conditions</italic>. This constraint not only improved consistency but also aligned model outputs with clinically interpretable categories.</p></sec></sec><sec id="s2-7"><title>Alternative Framework Compositions</title><p>We also evaluated 3 alternative framework compositions: F1 included the Short-CoT agent and the multiple-choice constraint agent but omitted the summary agent; F2 included the guideline-derived long CoT (Long-CoT) agent and the multiple-choice constraint agent but omitted the summary agent; F3 included the summary agent and the Long-CoT agent combined with the multiple-choice constraint agent. This part used a Long-CoT agent to augment the selected LLM decision-making. The Long-CoT involved filtering and restructuring clinical guidelines into a structured decision-making pathway for AIS (refer to the prompt provided in Supplementary Material Methods in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) [<xref ref-type="bibr" rid="ref23">23</xref>].</p></sec><sec id="s2-8"><title>Model-Level Performance and Output Safety Assessment</title><p>Ground truth for model-level evaluation was defined as the actual treatment decision and corresponding TOAST classification documented in the clinical record, with independent verification for guideline concordance performed by 2 senior stroke clinicians: a neurologist with &#x003E;10 years of experience and a neurointerventional physician with &#x003E;15 years of experience. Cases with disagreement or ambiguity regarding guideline concordance were jointly re-evaluated, and eligibility for inclusion was determined by consensus.</p><p>The model-level primary outcome was the overall therapeutic and TOAST diagnostic accuracy of LLMs. Accuracy was the case-level proportion correct; TOAST F1 was macroaveraged one-vs-rest F1 across classes, and diagnostic F1 was computed on binary correctness (1/0). Model-level secondary outcomes encompassed safety-related assessments, including instruction adherence, structured harmfulness assessment, and other exploratory safety metrics. For each case, the presence of any omission or hallucination was counted as one event (yes or no), and event rates were computed as events per case (events/total cases) [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>].</p></sec><sec id="s2-9"><title>Comprehensive Multicenter, Multisource, and Cross-Scenario Evaluation</title><sec id="s2-9-1"><title>Validation Using PubMed Case Reports and External Cohorts</title><p>To clinically validate the framework, we analyzed real-world cases from group B using 6 open-source LLMs. To further test generalizability across model families, we also evaluated 6 open-source LLMs and an additional closed-source LLM in group C (PubMed case reports). The PubMed case report dataset is publicly accessible and predominantly comprises diagnostically challenging cases. This dataset therefore provided a more stringent test of the generalizability of the multiagent framework across patients with varying levels of difficulty and complexity.</p></sec><sec id="s2-9-2"><title>Prospective Clinical Validation With Multilevel and Multispecialty Physicians</title><p>To evaluate the clinical impact of LLM assistance, we conducted a prospective, blinded physician study at center A (approval 2024-KY-203) between February and May 2025. Twelve physicians with heterogeneous AIS expertise were recruited across 4 Chinese cities or provinces (Hunan, Guangdong, Jilin, and Shanghai), comprising junior (n=5), senior (n=5), and expert strata (n=2) and including stroke specialists and nonspecialists. Consecutive eligible patients were enrolled and randomized at enrollment to an AI-assisted arm (with LLM support) or a standard review arm (without LLM support). Each case was evaluated only once by a single assigned physician.</p><p>In the AI-assisted arm, the best-performing model from the retrospective evaluation (DeepSeek-R1) generated treatment recommendations, TOAST classifications, and structured reasoning. These model outputs were provided to physicians together with the routinely available clinical materials for each assigned case. Physicians additionally provided a 5-point Likert rating of the perceived effectiveness of the AI-assisted reasoning (with higher scores indicating greater perceived usefulness). In the standard review arm, physicians reviewed identical case materials without any model output or additional computer-based decision support beyond routine hospital systems.</p><p>The reference standard was the final multidisciplinary team decision. Additional design details, including sample-size estimation and allocation procedures, are reported in the Supplementary Material Methods in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec></sec><sec id="s2-10"><title>Statistical Analysis</title><p>All analyses were performed in Python (version 3.10; Python Software Foundation). Statistical significance was set at a 2-sided <italic>P</italic>&#x003C;.05. Binary and continuous variables were summarized descriptively. Accuracy differences between standalone and LLMs augmented by the multiagent framework were tested with the McNemar test, and <italic>F</italic><sub>1</sub>-score differences were estimated by bootstrap resampling (1000 iterations) with 95% CIs. Paired ordinal outcomes were compared using the Wilcoxon signed-rank test. The ablation study compared the full multiagent framework with predefined variants that removed or substituted key modules (summary agent, guideline-derived Long-CoT vs Short-CoT, and free-form vs constrained multiple-choice outputs) [<xref ref-type="bibr" rid="ref26">26</xref>]. We fitted physician-case&#x2013;level binomial-logit generalized linear mixed-effects models (GLMMs), with decision correctness as the binary outcome, and LLM support as the main fixed effect for treatment decision and TOAST classification [<xref ref-type="bibr" rid="ref27">27</xref>]. Event rates&#x2014;including omission and hallucination frequencies&#x2014;and differences between AI-assisted and nonassisted arms were analyzed using chi-square or Fisher exact tests, as appropriate. Outcome metrics and definitions are summarized in Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Patient Characteristics</title><p>At center A, 1228 cases were screened retrospectively, and 1055 (85.91%) were included in group A. At center B, 938 were screened, and 721 (76.87%) were included in group B. In addition, 213 prospective cases were screened at center A between February and May 2025, of which 161 (75.59%) were included in group D. Of 327 PubMed case reports retrieved, 144 (44.04%) met the inclusion criteria and were included in group C. We finally analyzed 1937 clinical cases (mean age 68.0, SD 13.6 years; n=761, 39.29% women) between January 2018 and May 2025, and 144 PubMed case reports identified from January 2024 to January 2025 (<xref ref-type="table" rid="table1">Table 1</xref>).</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Characteristics of enrolled patients.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Variables</td><td align="left" valign="bottom">Group A (center A; n=1055)</td><td align="left" valign="bottom">Group B (center B; n=721)</td><td align="left" valign="bottom">Group C (PubMed cases; n=144)</td><td align="left" valign="bottom">Group D (center A; n=161)</td></tr></thead><tbody><tr><td align="left" valign="top">Age (y), mean (SD)</td><td align="left" valign="top">69.86 (13.26)</td><td align="left" valign="top">66.55 (11.76)</td><td align="left" valign="top">56.46 (18.73)</td><td align="left" valign="top">72.32 (12.22)</td></tr><tr><td align="left" valign="top">Male, n (%)</td><td align="left" valign="top">665 (63.03)</td><td align="left" valign="top">485 (67.27)</td><td align="left" valign="top">71<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> (50)</td><td align="left" valign="top">97 (60.25)</td></tr><tr><td align="left" valign="top">Female, n (%)</td><td align="left" valign="top">390 (36.97)</td><td align="left" valign="top">236 (32.73)</td><td align="left" valign="top">71<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> (50)</td><td align="left" valign="top">64 (39.75)</td></tr><tr><td align="left" valign="top" colspan="5">Disease categories, n (%)</td></tr><tr><td align="left" valign="top">&#x2003;AIS<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">998 (94.6)</td><td align="left" valign="top">610 (84.6)</td><td align="left" valign="top">127 (88.19)</td><td align="left" valign="top">159 (98.76)</td></tr><tr><td align="left" valign="top">&#x2003;Cerebral hemorrhage</td><td align="left" valign="top">26 (2.46)</td><td align="left" valign="top">64 (8.88)</td><td align="left" valign="top">1 (0.69)</td><td align="left" valign="top">1 (0.62)</td></tr><tr><td align="left" valign="top">&#x2003;Epilepsy</td><td align="left" valign="top">8 (0.76)</td><td align="left" valign="top">18 (2.5)</td><td align="left" valign="top">1 (0.69)</td><td align="left" valign="top">0 (0)</td></tr><tr><td align="left" valign="top">&#x2003;Arterial aneurysm</td><td align="left" valign="top">4 (0.38)</td><td align="left" valign="top">3 (0.42)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">1 (0.62)</td></tr><tr><td align="left" valign="top">&#x2003;Transient ischemic attack</td><td align="left" valign="top">9 (0.85)</td><td align="left" valign="top">22 (3.05)</td><td align="left" valign="top">4 (2.78)</td><td align="left" valign="top">0 (0)</td></tr><tr><td align="left" valign="top">&#x2003;Other non-AIS diseases</td><td align="left" valign="top">10 (0.95)</td><td align="left" valign="top">4 (0.55)</td><td align="left" valign="top">11 (7.64)</td><td align="left" valign="top">0 (0)</td></tr><tr><td align="left" valign="top" colspan="5">AIS treatment n (%)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td></tr><tr><td align="left" valign="top">&#x2003;Thrombolysis</td><td align="left" valign="top">135 (13.53)</td><td align="left" valign="top">135 (22.13)</td><td align="left" valign="top">17 (13.39)</td><td align="left" valign="top">20 (12.58)</td></tr><tr><td align="left" valign="top">&#x2003;Endovascular thrombectomy<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td><td align="left" valign="top">297 (29.76)</td><td align="left" valign="top">125 (20.49)</td><td align="left" valign="top">44 (34.56)</td><td align="left" valign="top">27 (16.98)</td></tr><tr><td align="left" valign="top">&#x2003;Standard medical management</td><td align="left" valign="top">565 (56.61)</td><td align="left" valign="top">350 (57.38)</td><td align="left" valign="top">66 (51.97)</td><td align="left" valign="top">112 (70.44)</td></tr><tr><td align="left" valign="top" colspan="5">TOAST<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup> diagnosis of AIS n (%)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td></tr><tr><td align="left" valign="top">&#x2003;Large artery atherosclerosis</td><td align="left" valign="top">760 (76.15)</td><td align="left" valign="top">494 (80.98)</td><td align="left" valign="top">41 (32.28)</td><td align="left" valign="top">99 (62.26)</td></tr><tr><td align="left" valign="top">&#x2003;Cardioembolism</td><td align="left" valign="top">98 (9.82)</td><td align="left" valign="top">40 (6.56)</td><td align="left" valign="top">28 (22.04)</td><td align="left" valign="top">21 (13.21)</td></tr><tr><td align="left" valign="top">&#x2003;Small vessel occlusion</td><td align="left" valign="top">95 (9.52)</td><td align="left" valign="top">37 (6.07)</td><td align="left" valign="top">6 (4.7)</td><td align="left" valign="top">37 (23.27)</td></tr><tr><td align="left" valign="top">&#x2003;Other causes</td><td align="left" valign="top">26 (2.61)</td><td align="left" valign="top">15 (2.46)</td><td align="left" valign="top">45 (35.43)</td><td align="left" valign="top">1 (0.63)</td></tr><tr><td align="left" valign="top">&#x2003;Cryptogenic</td><td align="left" valign="top">18 (1.80)</td><td align="left" valign="top">24 (3.93)</td><td align="left" valign="top">5 (3.94)</td><td align="left" valign="top">1 (0.63)</td></tr><tr><td align="left" valign="top">NIHSS<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup>, mean (SD)</td><td align="left" valign="top">8.3 (9.1)</td><td align="left" valign="top">6.3 (7.1)</td><td align="left" valign="top">6.9 (8.3)</td><td align="left" valign="top">7.1 (8.3)</td></tr><tr><td align="left" valign="top">Time from symptom onset (min), median (IQR)</td><td align="left" valign="top">360.0 (180-1440)</td><td align="left" valign="top">360.0 (150-1435)</td><td align="left" valign="top">660.0 (210-1440)</td><td align="left" valign="top">360.0 (210-960)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Two cases had missing sex information.</p></fn><fn id="table1fn2"><p><sup>b</sup>AIS: acute ischemic stroke.</p></fn><fn id="table1fn3"><p><sup>c</sup>Percentages for AIS treatment and TOAST diagnosis were calculated using the number of confirmed AIS cases in each group as the denominator (Group A: n=998; Group B: n=610; Group C: n=127; Group D: n=159).</p></fn><fn id="table1fn4"><p><sup>d</sup>Including bridging therapy (thrombolysis and endovascular thrombectomy).</p></fn><fn id="table1fn5"><p><sup>e</sup>TOAST: Trial of ORG 10172 in Acute Stroke Treatment.</p></fn><fn id="table1fn6"><p><sup>f</sup>NIHSS: National Institutes of Health Stroke Scale.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Baseline Performance Disparities Among Standalone LLMs</title><p><xref ref-type="fig" rid="figure2">Figure 2A and 2B</xref> and Figure S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> illustrate the simplified system architecture for the standalone LLM and the LLMs augmented by the multiagent framework. Performance varied markedly by model size. Standalone larger-scale LLMs consistently outperformed smaller ones in both treatment recommendation and TOAST classification (<xref ref-type="table" rid="table2">Table 2</xref>). Additionally, the multiagent framework consistently improved macroaveraged sensitivity and specificity across all evaluated models (Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). For example, across the pooled cohort, GPT-OSS-120B achieved 0.724 accuracy for treatment recommendation, compared with 0.570 for Baichuan-M1-14B and 0.464 for GPT-OSS-20B (all adjusted <italic>P</italic>&#x003C;.0001; <xref ref-type="table" rid="table2">Table 2</xref>). Performance trends were consistent across individual subgroups (Table S4-S7 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Similar gaps were observed for TOAST classification, with DeepSeek-R1 surpassing all smaller models. These results establish model size as a key determinant of baseline performance.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Simplified system architecture and performance of standalone vs framework-augmented large language models (LLMs). (A) and (B) Simplified system architecture: constrained inputs use &#x003C;think&#x003E;, &#x003C;treatment&#x003E;, and &#x003C;diagnosis&#x003E; tags. (C-H) Accuracy of acute ischemic stroke (AIS) treatment recommendation for 6 LLMs (Baichuan-M1-14B, GPT-OSS-20B, Qwen2.5-32B, DeepSeek-R1-Distill-Qwen-32B, GPT-OSS-120B, and DeepSeek-R1-671B) across groups A-C; paired bars compare standalone with multiagent, showing consistent gains. In group C, GPT-4o gained accuracy in treatment recommendation (+14.9<bold>%,</bold> 0.750 vs 0.653). TOAST: Trial of ORG 10172 in Acute Stroke Treatment. *<italic>P</italic>&#x003C;.05, **<italic>P</italic>&#x003C;.01, ***<italic>P</italic>&#x003C;.001.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e96304_fig02.png"/></fig><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Accuracy of large language models (LLMs) for treatment recommendation and Trial of ORG 10172 in Acute Stroke Treatment (TOAST) classification in the pooled group.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Models and accuracy<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td><td align="left" valign="bottom">Standalone LLM, accuracy</td><td align="left" valign="bottom">Multiagent framework, accuracy</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="4">Baichuan-M1-14B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Treatment</td><td align="left" valign="top">0.570 (0.547-0.591)</td><td align="left" valign="top">0.695 (0.675-0.715)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>TOAST</td><td align="left" valign="top">0.623 (0.602-0.645)</td><td align="left" valign="top">0.657 (0.637-0.677)</td><td align="left" valign="top">.10</td></tr><tr><td align="left" valign="top" colspan="4">GPT-OSS-20B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Treatment</td><td align="left" valign="top">0.464 (0.441-0.486)</td><td align="left" valign="top">0.541 (0.519-0.56)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>TOAST</td><td align="left" valign="top">0.530 (0.507-0.553)</td><td align="left" valign="top">0.623 (0.602-0.645)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top" colspan="4">Qwen2.5-32B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Treatment</td><td align="left" valign="top">0.593 (0.574-0.615)</td><td align="left" valign="top">0.693 (0.674-0.713)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>TOAST</td><td align="left" valign="top">0.645 (0.622-0.666)</td><td align="left" valign="top">0.706 (0.686-0.726)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top" colspan="4">DeepSeek-R1-Distill-Qwen-32B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Treatment</td><td align="left" valign="top">0.599 (0.577-0.620)</td><td align="left" valign="top">0.734 (0.716-0.754)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>TOAST</td><td align="left" valign="top">0.643 (0.622-0.663)</td><td align="left" valign="top">0.689 (0.669-0.708)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top" colspan="4">GPT-OSS-120B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Treatment</td><td align="left" valign="top">0.724 (0.706-0.744)</td><td align="left" valign="top">0.825 (0.808-0.840)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>TOAST</td><td align="left" valign="top">0.690 (0.668-0.710)</td><td align="left" valign="top">0.711 (0.690-0.732)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top" colspan="4">DeepSeek-R1-671B</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Treatment</td><td align="left" valign="top">0.685 (0.667-0.704)</td><td align="left" valign="top">0.830 (0.813-0.847)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>TOAST</td><td align="left" valign="top">0.680 (0.659-0.701)</td><td align="left" valign="top">0.758 (0.738-0.777)</td><td align="left" valign="top">&#x003C;.001</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Values are presented as point estimates, with 95% confidence intervals shown in parentheses.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Multiagent Framework-Induced Improvements Across LLMs</title><p>Augmentation with the multiagent framework substantially improved outcomes across all models (<xref ref-type="fig" rid="figure2">Figure 2C-2H</xref> and <xref ref-type="fig" rid="figure3">Figure 3A</xref>; Figure S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), with an average improvement of 18.9% compared with standalone LLMs. In group A, Baichuan-M1-14B accuracy increased by 25.8% (<italic>F</italic><sub>1</sub>-score +15.1%), while DeepSeek-R1-671B improved more modestly (+23.3% accuracy and <italic>F</italic><sub>1</sub>-score +12.7%). For TOAST classification, GPT-OSS-20B accuracy rose by 39.0% (<italic>F</italic><sub>1</sub>-score +4.9%), compared with only 2.5% (<italic>F</italic><sub>1</sub>-score &#x2212;2.1%) for GPT-OSS-120B. Notably, GPT-OSS-20B exhibited unstable outputs with fluctuating gains. Collectively, these findings show that while model size remains critical for baseline accuracy, the multiagent framework reduces size-related disparities, enabling smaller models to approach the clinical utility of their larger counterparts.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p><italic>F</italic><sub>1</sub>-score comparison of large language models (LLMs) in acute ischemic stroke (AIS) treatment recommendation and Trial of ORG 10172 in Acute Stroke Treatment (TOAST) classification (A) and treatment recommendation flows (B)<bold>.</bold> (A) <italic>F</italic><sub>1</sub>-scores are shown for AIS treatment recommendation and TOAST classification across groups A-C and 6 LLMs of increasing scale. Within each LLM, paired dots represent standalone performance (orange) and framework-augmented LLMs (blue), with error bars indicating CIs. The <italic>F</italic><sub>1</sub>-score highlights the enhanced diagnostic reliability of the multiagent framework. (B) Sankey diagram comparing treatment recommendation flows between the standalone and the multiagent framework. Asterisks denote within-group differences between conditions, 2-sided paired tests. *<italic>P</italic>&#x003C;.05, **<italic>P</italic>&#x003C;.01, ***<italic>P</italic>&#x003C;.001.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e96304_fig03.png"/></fig></sec><sec id="s3-4"><title>Cross-Group and Multisource Validation</title><p>Validation in groups B and C demonstrated consistent performance gains across all groups (<xref ref-type="fig" rid="figure2">Figure 2C-2H</xref> and <xref ref-type="fig" rid="figure3">Figure 3A</xref>; Tables S4-S7 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The framework-augmented DeepSeek-R1-671B achieved the highest accuracy in both treatment recommendation and TOAST classification. In group C, GPT-4o gained accuracy in treatment recommendation (+14.9%, 0.750 vs 0.653) but showed a decline in TOAST classification (&#x2212;6.7%, 0.486 vs 0.521), reflecting the predominance of &#x201C;other causes&#x201D; and cryptogenic subtypes. The results for alternative framework compositions are shown in <xref ref-type="fig" rid="figure4">Figure 4</xref>. Collectively, these findings confirm that the framework provides reliable gains across heterogeneous cohorts.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Performance of standalone large language models (LLMs), 3 alternative framework compositions, and our multiagent framework<bold>.</bold> Panels A-F show accuracy for acute ischemic stroke (AIS) treatment recommendation across groups A-C. Paired bars summarize accuracy profiles across settings, providing a direct comparison of different framework compositions<italic>.</italic></p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e96304_fig04.png"/></fig></sec><sec id="s3-5"><title>Contribution of Framework Components</title><p>Generalized linear mixed model analysis in group A confirmed that the multiagent framework was independently associated with higher accuracy in AIS treatment recommendations (odds ratio [OR] 1.72, 95% CI 1.63-1.81; <italic>P</italic>&#x003C;.001) and increased selection of appropriate reperfusion strategies (<xref ref-type="fig" rid="figure3">Figure 3B</xref>).</p><p>In the ablation study (average-only aggregation), treatment recommendation performance improved when using a multiple-choice constraint compared with free-form answer formats (constrained vs free-form: +0.038), and CoT prompting outperformed direct prompting (long vs direct: +0.052; short vs direct: +0.063), suggesting that treatment recommendations benefit from structured answer formats and explicit reasoning scaffolds. Summarization yielded only a modest average gain for treatment recommendation (summary vs no summary: +0.006). In contrast, diagnosis benefited more from summarization (+0.041) and performed better with free-form answer formats (constrained vs free-form: &#x2212;0.051), motivating task-specific choices of prompting strategy and answer-format control.</p></sec><sec id="s3-6"><title>Qualifying LLM Reasoning for Safe Clinical Deployment</title><p>Beyond accuracy, we visualized step-by-step reasoning traces for representative success and failure cases to characterize failure modes and deployment-relevant clinical safety risks (<xref ref-type="fig" rid="figure5">Figure 5A and 5B</xref>). Because accuracy alone provides an incomplete view of clinical applicability, we further assessed output safety. Compared with the standalone LLM, the multiagent framework achieved higher mean clinical safety scores (4.01, SD 0.35 vs 3.70, SD 0.43) and lower mean hallucination (20.6% vs 33.6%) and omission rates (24.5% vs 38.5%; <xref ref-type="fig" rid="figure5">Figure 5C-5E</xref>; Table S8 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Other exploratory safety metrics are shown in Table S9 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. These findings demonstrate that structured reasoning may improve reliability and mitigate the risk of unsafe content entering clinical workflows.</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Case examples and safety profile of large language model (LLM) outputs. (A-B), Representative cases. (A) Correct recommendation with faithful reasoning and no hallucination or omission. (B) Incorrect recommendation: relevant details were identified, but hallucinated content led to an erroneous conclusion. (C-E) Safety metrics across 6 LLMs. (C) Harmfulness ratings on a 3-point scale (1=not harmful; 3=highly harmful) shown as a concentric doughnut plot with overlaid points. (D) Instruction-following compliance is shown as a bar plot of adherence proportions. (E) Incidence rates of hallucination and omission are shown as line plots. <italic>*P&#x003C;.05, **P&#x003C;.01, ***P&#x003C;.001.</italic></p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e96304_fig05.png"/></fig></sec><sec id="s3-7"><title>LLM Support Benefits Less-Experienced Physicians</title><p>Integration of LLM support substantially enhanced physician performance, with the most pronounced gains observed for less-experienced physicians in treatment decisions (0.667 to 0.833 in junior specialists; 0.600 to 0.846 in nonspecialists). On the basis of physician-case decision-level observations (1 observation per physician per case), we fitted a binomial-logit GLMM with decision correctness as the binary outcome, LLM support, physician experience, and specialty as fixed effects, and a random intercept for case and physician. After adjustment, LLM support was associated with higher odds of correct treatment decisions (OR 2.86, 95% CI 2.27&#x2010;3.60; <italic>P</italic>&#x003C;.0001) and correct TOAST classification (OR 3.63, 95% CI 2.85&#x2010;4.64; <italic>P</italic>&#x003C;.0001; Figure S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The estimated association was strongest among junior and nonspecialist physicians. For treatment decisions, the ORs were 3.96 (95% CI 2.64-5.96) for junior nonspecialists and 3.22 (95% CI 2.13-4.87) for senior nonspecialists; for TOAST classification, the corresponding ORs were 4.66 (95% CI 3.01-7.22) and 4.27 (95% CI 2.74-6.66). Expert specialists showed positive but nonsignificant estimates, consistent with a ceiling effect at higher baseline performance levels (<italic>P</italic>&#x003C;.001). Physicians further rated the assistance positively, with a mean score of 3.849 out of 5. Improvements were more modest in senior and specialist physicians (<xref ref-type="fig" rid="figure6">Figure 6</xref>).</p><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Human-AI evaluation across physician groups<bold>.</bold> (A) Treatment recommendation accuracy without AI (pink) and with AI (blue); bars show means with 95% CIs. (B) Individual physicians&#x2019; treatment <italic>F</italic><sub>1</sub>-scores; lines connect the same physician from &#x201C;without AI&#x201D; to &#x201C;with AI.&#x201D; (C) Trial of ORG 10172 in Acute Stroke Treatment (TOAST) classification accuracy by physician group, as in panel A. (D) Individual physicians&#x2019; TOAST <italic>F</italic><sub>1</sub>-scores, as in panel B. (E) Perceived effectiveness of AI-assisted reasoning on a 5-point Likert scale (5=highly useful or transformative; 1=harmful or ineffective), stratified by physician group. (F) Radar plots summarizing accuracy profiles for treatment recommendation (left) and TOAST classification (right), with and without AI assistance. Physician groups are stratified by seniority (junior vs senior, including both specialists and nonspecialists) and by specialty background (specialist vs nonspecialist, including both junior and senior physicians)<italic>. *P&#x003C;.05, **P&#x003C;.01, ***P&#x003C;.001.</italic></p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e96304_fig06.png"/></fig></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><p>This study advances the clinical translation of LLMs for AIS decision support by moving beyond benchmark-style evaluation to a multicenter, multisource assessment across retrospective, prospective, and literature-derived cases. First, the structured multiagent framework improved guideline-concordant therapy selection while producing more reliable and clinically accountable outputs, with lower hallucination and omission rates. Second, in a human-AI interaction experiment involving physicians of varying seniority and specialty backgrounds, the framework improved physician decision accuracy, with the largest gains observed among junior and nonspecialist physicians. Collectively, these findings suggest that LLM-integrated systems can enhance both intrinsic decision reliability and downstream clinical performance, offering a practical means to improve consistency, safety, and equity in AIS care when access to stroke expertise is uneven.</p><p>Global inequities in stroke care remain profound, with low- and middle-income countries constrained by limited resources and specialist expertise, while high-income countries face overcrowded emergency services and workforce pressures. Existing evaluations of LLMs have been largely confined to simplified benchmark tasks that fail to capture the real-world complexity of disease management [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>]. To address this gap, we applied LLMs to real-world clinical and literature-derived cases, thereby simulating authentic clinical scenarios and reflecting heterogeneous contexts. Performance declined in complex settings, partly due to limitations in handling long token inputs [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]. To address this challenge, our multiagent framework incorporated a custom-designed summarization agent that extracted salient features from clinical narratives, improving accuracy across diverse scenarios. Augmented LLMs achieved accuracies ranging from 0.541 to 0.830 across model sizes&#x2014;a level comparable to question-and-answer&#x2013;style or simulated patient scenarios [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref30">30</xref>], and consistently higher than standalone LLMs.</p><p>This study represents an initial step toward translating LLM-assisted decision support for AIS into clinical practice, with a focus on reducing variability in care that arises from uneven clinical expertise. Prior work suggests that CoT prompting and fixed-answer formats can partially address these limitations [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref32">32</xref>]. Our multiagent framework introduces a workflow-oriented, guideline-concordant structure that improves decision accuracy while maintaining clinically consistent recommendations, supporting feasibility for future clinical integration. Unlike approaches that rely on ever-larger proprietary models [<xref ref-type="bibr" rid="ref33">33</xref>], our results show that workflow-inspired structured design improvements can yield substantial gains without increasing model scale, potentially lowering adoption barriers and supporting broader access to specialist-level decision support.</p><p>Our evaluation offers one of the most comprehensive simulations of clinical practice to date, establishing a foundation for the deployment of LLMs in AIS workflows. While the promise is substantial, deployment at scale carries risks of unintended harmful consequences [<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref35">35</xref>]. By integrating data from multicenter, multitier hospitals, retrospective and prospective real-world cases, literature-derived high-difficulty cases, and human-AI interactions across physicians of different levels and specialties, our evaluation captured the heterogeneity and complexity of AIS care. Notably, LLMs provided the most benefit to less-experienced physicians, narrowing expertise gaps across experience, specialty, and geography, and aligning with prior reports of near expert-level performance [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>]. In alignment with the World Stroke Organization&#x2019;s Global Stroke Declaration, which underscores that quality stroke care should be universal [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref38">38</xref>], our results advance the case for AI-enabled strategies to promote equity in global stroke systems.</p><p>This study systematically assessed safety, a critical prerequisite for clinical deployment. Because LLMs predict the next token without verifying evidence, they remain prone to hallucinations that can erode trust and generate harmful or misleading recommendations [<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref40">40</xref>]. In our evaluation, hallucinations and omissions were not eliminated but occurred at relatively low frequencies, with the multiagent DeepSeek-R1-671B achieving a hallucination rate of 10.9%, an omission rate of 14.7%, and an overall clinical safety score of 4.36 out of 5. These findings indicate that while safety concerns remain, the error rates are within a range that may be acceptable for decision support use, supporting the feasibility of cautious clinical integration [<xref ref-type="bibr" rid="ref41">41</xref>]. The multiagent-augmented LLMs further demonstrated stronger instruction adherence, lower hallucination and omission rates, and more reliable structured outputs, thereby addressing a key barrier to real-world deployment.</p><p>Our results suggest that structured multiagent LLM frameworks may improve guideline-concordant AIS decision support, output safety, and auditability, but the remaining safety-critical errors highlight the need for caution before clinical deployment. Such systems should be used only as human-in-the-loop decision-support tools, with final treatment decisions remaining under physician responsibility. Future workflow-based studies should prospectively test safeguards, including uncertainty signaling, guideline and local-protocol alignment, contraindication review, deferral under missing or ambiguous information, high-risk warnings, and complete logging of model inputs, outputs, and reasoning traces.</p><p>This study has several limitations. First, the framework was evaluated in case-based rather than real-time emergency room settings, precluding assessment of usability, clinician trust, latency, time to decision, bedside integration, patient outcomes, and uncertainty escalation. Second, the exclusion criteria may have produced a more standardized dataset than real-world emergency stroke care. By excluding cases with incomplete information, chronic-phase presentations, competing urgent conditions, or treatment refusal or discontinuation, the study may underrepresent information-limited and operationally constrained scenarios. Third, although the framework reduced omission and hallucination events, the proposed deployment guardrails were not prospectively tested in real clinical workflows, limiting conclusions about operational safety. Fourth, the automated grader used to categorize free-text standalone LLM outputs may have introduced evaluation error, particularly for ambiguous or internally inconsistent responses. Finally, rapid iteration of commercial LLMs and restricted access to proprietary systems may affect reproducibility and long-term stability. Future prospective multicenter workflow studies should evaluate the framework under more complex real-world conditions, with predefined safety guardrails, human adjudication of uncertain outputs, and patient-level outcome assessment.</p><p>In conclusion, our multicenter evaluation demonstrates that a structured, multiagent LLM framework significantly enhances guideline-concordant AIS decision-making (average accuracy gain: 18.9%) and output safety. Crucially, LLM support substantially increased the odds of correct physician decisions for both treatment (OR 2.86, 95% CI 2.27&#x2010;3.60) and TOAST classification (OR 3.63, 95% CI 2.85&#x2010;4.64), with the most pronounced benefits among less-experienced and nonspecialist clinicians. These findings highlight the framework&#x2019;s potential to narrow expertise gaps and support equitable stroke care, warranting further prospective evaluations of real-world workflows and patient outcomes before clinical deployment.</p></sec></body><back><ack><p>This study was conducted over an extended period and required substantial human and material resources. The authors thank all patients and their families for their participation, as well as the open-source community for making large language models (LLMs) publicly available. The authors are grateful to their cooperators for assistance with data collection and coding, and to the biostatistics experts for their valuable guidance on statistical methodology. They also acknowledge the strong support in clinical validation from Tao Wang, Hongmei Song, Daqian Zhang, Yingying Lu, Tonglei Fang, Xingxing Sun, Lu Fei, Yixiao Tang, Yifan Tu, Zhongzheng Cao, Fasheng Peng, Mengfan Yan, and Yuxiang Zhou.</p><p>During the preparation of this manuscript, generative AI tools, including ChatGPT (OpenAI), were used solely for language editing and linguistic polishing to improve clarity and readability. No generative AI tools were used for study design, data collection, data analysis, interpretation of results, or generation of scientific content. All scientific content, results, and conclusions were developed independently by the authors, who take full responsibility for the integrity and accuracy of the manuscript.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the National Natural Science Foundation of China (8225024), the Key R&#x0026;D subproject of the Ministry of Science and Technology (2023YFF1204804 and No. 2023YFF1204804), the Shanghai Pudong New Area Science and Technology Commission Project (PKJ2023-Y53), the Shanghai Jiaotong University, Medicine and engineering interdisciplinary program (YG2024LC08), the Shanghai key discipline of medical imaging (2017ZZ02005 and No. 2024ZZ1011), and Drug and instrument program of Shanghai Science and Technology (24SF1903900). The funders had no involvement in the study design, data collection, analysis, interpretation of results, or writing of the manuscript.</p></sec><sec><title>Data Availability</title><p>The raw data supporting the findings of this study are available from the corresponding author upon reasonable request. For the PubMed case cohort, the original data are not directly shared in this work; instead, we provide the references to the corresponding open-access publications in our released code repository, from which the source data can be obtained.</p><p>All code for this study is publicly available. The source code for model deployment, inference scripts, prompts, and trained model weights used in this work are available in the released code repository [<xref ref-type="bibr" rid="ref42">42</xref>]. The following LLMs were evaluated: Baichuan-M1-14B, GPT-OSS-20B, Qwen2.5-32B, DeepSeek-R1-Distill-Qwen-32B, GPT-OSS-120B, DeepSeek-R1-671B, and GPT-4o.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: BY</p><p>Data curation: BY, XS, ZC</p><p>Formal analysis: RZ</p><p>Funding acquisition: YL</p><p>Investigation: BY, LC, XS</p><p>Methodology: BY, RZ, LC</p><p>Project administration: BY</p><p>Software: RZ</p><p>Supervision: YL</p><p>Validation: ZC, LC</p><p>Writing&#x2014;original draft: BY</p><p>Writing&#x2014;review and editing: BY, RZ, YL</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AIS</term><def><p>acute ischemic stroke</p></def></def-item><def-item><term id="abb2">CoT</term><def><p>chain of thought</p></def></def-item><def-item><term id="abb3">GLMM</term><def><p>generalized linear mixed-effects model</p></def></def-item><def-item><term id="abb4">GPU</term><def><p>graphics processing unit</p></def></def-item><def-item><term id="abb5">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb6">Long-CoT</term><def><p>guideline-derived long chain of thought</p></def></def-item><def-item><term id="abb7">OR</term><def><p>odds ratio</p></def></def-item><def-item><term id="abb8">Short-CoT</term><def><p>reasoning-path chain of thought</p></def></def-item><def-item><term id="abb9">TOAST</term><def><p>Trial of ORG 10172 in Acute Stroke Treatment</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><collab>GBD 2021 Forecasting Collaborators</collab></person-group><article-title>Burden of disease scenarios for 204 countries and territories, 2022-2050: a forecasting analysis for the Global Burden of Disease Study 2021</article-title><source>Lancet</source><year>2024</year><month>05</month><day>18</day><volume>403</volume><issue>10440</issue><fpage>2204</fpage><lpage>2256</lpage><pub-id pub-id-type="doi">10.1016/S0140-6736(24)00685-8</pub-id><pub-id pub-id-type="medline">38762325</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Feigin</surname><given-names>VL</given-names> </name><name name-style="western"><surname>Brainin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Norrving</surname><given-names>B</given-names> </name><etal/></person-group><article-title>World Stroke Organization: Global Stroke Fact Sheet 2025</article-title><source>Int J Stroke</source><year>2025</year><month>02</month><volume>20</volume><issue>2</issue><fpage>132</fpage><lpage>144</lpage><pub-id pub-id-type="doi">10.1177/17474930241308142</pub-id><pub-id pub-id-type="medline">39635884</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shen</surname><given-names>YC</given-names> </name><name name-style="western"><surname>Sarkar</surname><given-names>N</given-names> </name><name name-style="western"><surname>Hsia</surname><given-names>RY</given-names> </name></person-group><article-title>Structural inequities for historically underserved communities in the adoption of stroke certification in the United States</article-title><source>JAMA Neurol</source><year>2022</year><month>08</month><day>1</day><volume>79</volume><issue>8</issue><fpage>777</fpage><lpage>786</lpage><pub-id pub-id-type="doi">10.1001/jamaneurol.2022.1621</pub-id><pub-id pub-id-type="medline">35759253</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pandian</surname><given-names>JD</given-names> </name><name name-style="western"><surname>Kalkonde</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Sebastian</surname><given-names>IA</given-names> </name><name name-style="western"><surname>Felix</surname><given-names>C</given-names> </name><name name-style="western"><surname>Urimubenshi</surname><given-names>G</given-names> </name><name name-style="western"><surname>Bosch</surname><given-names>J</given-names> </name></person-group><article-title>Stroke systems of care in low-income and middle-income countries: challenges and opportunities</article-title><source>Lancet</source><year>2020</year><month>10</month><day>31</day><volume>396</volume><issue>10260</issue><fpage>1443</fpage><lpage>1451</lpage><pub-id pub-id-type="doi">10.1016/S0140-6736(20)31374-X</pub-id><pub-id pub-id-type="medline">33129395</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Avasarala</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wesley</surname><given-names>K</given-names> </name></person-group><article-title>Optimization of acute stroke care in the emergency department: a call for better utilization of healthcare resources amid growing shortage of neurologists in the United States</article-title><source>CNS Spectr</source><year>2018</year><month>08</month><volume>23</volume><issue>4</issue><fpage>248</fpage><lpage>250</lpage><pub-id pub-id-type="doi">10.1017/S109285291700013X</pub-id><pub-id pub-id-type="medline">28209212</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Guo</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Burden of cardiovascular disease among the Western Pacific region and its association with human resources for health, 1990-2021: a systematic analysis of the Global Burden of Disease Study 2021</article-title><source>Lancet Reg Health West Pac</source><year>2024</year><volume>51</volume><fpage>101195</fpage><pub-id pub-id-type="doi">10.1016/j.lanwpc.2024.101195</pub-id><pub-id pub-id-type="medline">39286450</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nasreldein</surname><given-names>A</given-names> </name><name name-style="western"><surname>Walter</surname><given-names>S</given-names> </name><name name-style="western"><surname>Mohamed</surname><given-names>KO</given-names> </name><etal/></person-group><article-title>Pre- and in-hospital delays in the use of thrombolytic therapy for patients with acute ischemic stroke in rural and urban Egypt</article-title><source>Front Neurol</source><year>2022</year><volume>13</volume><fpage>1070523</fpage><pub-id pub-id-type="doi">10.3389/fneur.2022.1070523</pub-id><pub-id pub-id-type="medline">36742046</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Feigin</surname><given-names>VL</given-names> </name><name name-style="western"><surname>Roth</surname><given-names>GA</given-names> </name><name name-style="western"><surname>Naghavi</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Global burden of stroke and risk factors in 188 countries, during 1990-2013: a systematic analysis for the Global Burden of Disease Study 2013</article-title><source>Lancet Neurol</source><year>2016</year><month>08</month><volume>15</volume><issue>9</issue><fpage>913</fpage><lpage>924</lpage><pub-id pub-id-type="doi">10.1016/S1474-4422(16)30073-4</pub-id><pub-id pub-id-type="medline">27291521</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="web"><article-title>Global declaration on stroke</article-title><source>World Stroke Organization</source><year>2023</year><access-date>2025-09-18</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.world-stroke.org/news-and-blog/news/global-declaration-on-stroke-commitments-for-facing-stroke">https://www.world-stroke.org/news-and-blog/news/global-declaration-on-stroke-commitments-for-facing-stroke</ext-link></comment></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>G</given-names> </name><etal/></person-group><article-title>A generalist medical language model for disease diagnosis assistance</article-title><source>Nat Med</source><year>2025</year><month>03</month><volume>31</volume><issue>3</issue><fpage>932</fpage><lpage>942</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03416-6</pub-id><pub-id pub-id-type="medline">39779927</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kanjee</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Crowe</surname><given-names>B</given-names> </name><name name-style="western"><surname>Rodman</surname><given-names>A</given-names> </name></person-group><article-title>Accuracy of a generative artificial intelligence model in a complex diagnostic challenge</article-title><source>JAMA</source><year>2023</year><month>07</month><day>3</day><volume>330</volume><issue>1</issue><fpage>78</fpage><lpage>80</lpage><pub-id pub-id-type="doi">10.1001/jama.2023.8288</pub-id><pub-id pub-id-type="medline">37318797</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cabral</surname><given-names>S</given-names> </name><name name-style="western"><surname>Restrepo</surname><given-names>D</given-names> </name><name name-style="western"><surname>Kanjee</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Clinical reasoning of a generative artificial intelligence model compared with physicians</article-title><source>JAMA Intern Med</source><year>2024</year><month>05</month><day>1</day><volume>184</volume><issue>5</issue><fpage>581</fpage><lpage>583</lpage><pub-id pub-id-type="doi">10.1001/jamainternmed.2024.0295</pub-id><pub-id pub-id-type="medline">38557971</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Schaekermann</surname><given-names>M</given-names> </name><name name-style="western"><surname>Palepu</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Towards conversational diagnostic artificial intelligence</article-title><source>Nature</source><year>2025</year><month>06</month><volume>642</volume><issue>8067</issue><fpage>442</fpage><lpage>450</lpage><pub-id pub-id-type="doi">10.1038/s41586-025-08866-7</pub-id><pub-id pub-id-type="medline">40205050</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McDuff</surname><given-names>D</given-names> </name><name name-style="western"><surname>Schaekermann</surname><given-names>M</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Towards accurate differential diagnosis with large language models</article-title><source>Nature</source><year>2025</year><month>06</month><volume>642</volume><issue>8067</issue><fpage>451</fpage><lpage>457</lpage><pub-id pub-id-type="doi">10.1038/s41586-025-08869-4</pub-id><pub-id pub-id-type="medline">40205049</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Goh</surname><given-names>E</given-names> </name><name name-style="western"><surname>Gallo</surname><given-names>R</given-names> </name><name name-style="western"><surname>Hom</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Large language model influence on diagnostic reasoning: a randomized clinical trial</article-title><source>JAMA Netw Open</source><year>2024</year><month>10</month><day>1</day><volume>7</volume><issue>10</issue><fpage>e2440969</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.40969</pub-id><pub-id pub-id-type="medline">39466245</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Goh</surname><given-names>E</given-names> </name><name name-style="western"><surname>Gallo</surname><given-names>RJ</given-names> </name><name name-style="western"><surname>Strong</surname><given-names>E</given-names> </name><etal/></person-group><article-title>GPT-4 assistance for improvement of physician performance on patient care tasks: a randomized controlled trial</article-title><source>Nat Med</source><year>2025</year><month>04</month><volume>31</volume><issue>4</issue><fpage>1233</fpage><lpage>1238</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/39910272/?">https://pubmed.ncbi.nlm.nih.gov/39910272/?</ext-link></comment><pub-id pub-id-type="doi">10.1038/s41591-024-03456-y</pub-id><pub-id pub-id-type="medline">39910272</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kottlors</surname><given-names>J</given-names> </name><name name-style="western"><surname>Hahnfeldt</surname><given-names>R</given-names> </name><name name-style="western"><surname>G&#x00F6;rtz</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Large language models-supported thrombectomy decision-making in acute ischemic stroke based on radiology reports: feasibility qualitative study</article-title><source>J Med Internet Res</source><year>2025</year><month>02</month><day>13</day><volume>27</volume><fpage>e48328</fpage><pub-id pub-id-type="doi">10.2196/48328</pub-id><pub-id pub-id-type="medline">39946168</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>H</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>H</given-names> </name><name name-style="western"><surname>Song</surname><given-names>L</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>M</given-names> </name><name name-style="western"><surname>Cheng</surname><given-names>W</given-names> </name><etal/></person-group><article-title>Baichuan-M1: pushing the medical capability of large language models</article-title><source>arXiv</source><comment>Preprint posted online on  Feb 18, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2502.12671</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Agarwal</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ahmad</surname><given-names>L</given-names> </name><name name-style="western"><surname>Ai</surname><given-names>J</given-names> </name><name name-style="western"><surname>Altman</surname><given-names>S</given-names> </name><name name-style="western"><surname>Applebaum</surname><given-names>A</given-names> </name><name name-style="western"><surname>Arbus</surname><given-names>E</given-names> </name><etal/></person-group><article-title>gpt-oss-120b &#x0026; gpt-oss-20b model card</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 8, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2508.10925</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>A</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Hui</surname><given-names>B</given-names> </name><name name-style="western"><surname>Zheng</surname><given-names>B</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Qwen2.5 technical report</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 19, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2412.15115</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>D</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Song</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>P</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>Q</given-names> </name><etal/></person-group><article-title>DeepSeek-R1: incentivizing reasoning capability in LLMs via reinforcement learning</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 22, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2501.12948</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Hurst</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lerer</surname><given-names>A</given-names> </name><name name-style="western"><surname>Goucher</surname><given-names>AP</given-names> </name><name name-style="western"><surname>Perelman</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ramesh</surname><given-names>A</given-names> </name><name name-style="western"><surname>Clark</surname><given-names>A</given-names> </name><etal/></person-group><article-title>GPT-4o system card</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 25, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2410.21276</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Powers</surname><given-names>WJ</given-names> </name><name name-style="western"><surname>Rabinstein</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Ackerson</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Guidelines for the early management of patients with acute ischemic stroke: 2019 update to the 2018 guidelines for the early management of acute ischemic stroke: a guideline for healthcare professionals from the American Heart Association/American Stroke Association</article-title><source>Stroke</source><year>2019</year><month>12</month><volume>50</volume><issue>12</issue><fpage>e344</fpage><lpage>e418</lpage><pub-id pub-id-type="doi">10.1161/STR.0000000000000211</pub-id><pub-id pub-id-type="medline">31662037</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Omar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sorin</surname><given-names>V</given-names> </name><name name-style="western"><surname>Collins</surname><given-names>JD</given-names> </name><etal/></person-group><article-title>Multi-model assurance analysis showing large language models are highly vulnerable to adversarial hallucination attacks during clinical decision support</article-title><source>Commun Med (Lond)</source><year>2025</year><month>08</month><day>2</day><volume>5</volume><issue>1</issue><fpage>330</fpage><pub-id pub-id-type="doi">10.1038/s43856-025-01021-3</pub-id><pub-id pub-id-type="medline">40753316</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Asgari</surname><given-names>E</given-names> </name><name name-style="western"><surname>Monta&#x00F1;a-Brown</surname><given-names>N</given-names> </name><name name-style="western"><surname>Dubois</surname><given-names>M</given-names> </name><etal/></person-group><article-title>A framework to assess clinical safety and hallucination rates of LLMs for medical text summarisation</article-title><source>NPJ Digit Med</source><year>2025</year><month>05</month><day>13</day><volume>8</volume><issue>1</issue><fpage>274</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01670-7</pub-id><pub-id pub-id-type="medline">40360677</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>R</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Berlowitz</surname><given-names>D</given-names> </name><name name-style="western"><surname>Mez</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>H</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>H</given-names> </name></person-group><article-title>CARE-AD: a multi-agent large language model framework for Alzheimer&#x2019;s disease prediction using longitudinal clinical notes</article-title><source>NPJ Digit Med</source><year>2025</year><month>08</month><day>24</day><volume>8</volume><issue>1</issue><fpage>541</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01940-4</pub-id><pub-id pub-id-type="medline">40849361</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hedeker</surname><given-names>D</given-names> </name></person-group><article-title>A mixed-effects multinomial logistic regression model</article-title><source>Stat Med</source><year>2003</year><month>05</month><day>15</day><volume>22</volume><issue>9</issue><fpage>1433</fpage><lpage>1446</lpage><pub-id pub-id-type="doi">10.1002/sim.1522</pub-id><pub-id pub-id-type="medline">12704607</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>T</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>G</given-names> </name><name name-style="western"><surname>Do</surname><given-names>QD</given-names> </name><name name-style="western"><surname>Yue</surname><given-names>X</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>W</given-names> </name></person-group><article-title>Long-context LLMs struggle with long in-context learning</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 2, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2404.02060</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>G</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>Q</given-names> </name><etal/></person-group><article-title>Leveraging long context in retrieval augmented language models for medical question answering</article-title><source>NPJ Digit Med</source><year>2025</year><month>05</month><day>2</day><volume>8</volume><issue>1</issue><fpage>239</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01651-w</pub-id><pub-id pub-id-type="medline">40316710</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>F</given-names> </name><name name-style="western"><surname>Li</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Patient triage and guidance in emergency departments using large language models: multimetric study</article-title><source>J Med Internet Res</source><year>2025</year><month>05</month><day>15</day><volume>27</volume><fpage>e71613</fpage><pub-id pub-id-type="doi">10.2196/71613</pub-id><pub-id pub-id-type="medline">40374171</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Johri</surname><given-names>S</given-names> </name><name name-style="western"><surname>Jeong</surname><given-names>J</given-names> </name><name name-style="western"><surname>Tran</surname><given-names>BA</given-names> </name><etal/></person-group><article-title>An evaluation framework for clinical use of large language models in patient interaction tasks</article-title><source>Nat Med</source><year>2025</year><month>01</month><volume>31</volume><issue>1</issue><fpage>77</fpage><lpage>86</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03328-5</pub-id><pub-id pub-id-type="medline">39747685</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhong</surname><given-names>W</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Performance of ChatGPT-4o and four open-source large language models in generating diagnoses based on China's rare disease catalog: comparative study</article-title><source>J Med Internet Res</source><year>2025</year><month>06</month><day>18</day><volume>27</volume><fpage>e69929</fpage><pub-id pub-id-type="doi">10.2196/69929</pub-id><pub-id pub-id-type="medline">40532199</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Guan</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Integrated image-based deep learning and language models for primary diabetes care</article-title><source>Nat Med</source><year>2024</year><month>10</month><volume>30</volume><issue>10</issue><fpage>2886</fpage><lpage>2896</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03139-8</pub-id><pub-id pub-id-type="medline">39030266</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Habib</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>AL</given-names> </name><name name-style="western"><surname>Grant</surname><given-names>RW</given-names> </name></person-group><article-title>The Epic Sepsis Model falls short-the importance of external validation</article-title><source>JAMA Intern Med</source><year>2021</year><month>08</month><day>1</day><volume>181</volume><issue>8</issue><fpage>1040</fpage><lpage>1041</lpage><pub-id pub-id-type="doi">10.1001/jamainternmed.2021.3333</pub-id><pub-id pub-id-type="medline">34152360</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wong</surname><given-names>A</given-names> </name><name name-style="western"><surname>Otles</surname><given-names>E</given-names> </name><name name-style="western"><surname>Donnelly</surname><given-names>JP</given-names> </name><etal/></person-group><article-title>External validation of a widely implemented proprietary sepsis prediction model in hospitalized patients</article-title><source>JAMA Intern Med</source><year>2021</year><month>08</month><day>1</day><volume>181</volume><issue>8</issue><fpage>1065</fpage><lpage>1070</lpage><pub-id pub-id-type="doi">10.1001/jamainternmed.2021.2626</pub-id><pub-id pub-id-type="medline">34152373</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Owens</surname><given-names>D</given-names> </name><name name-style="western"><surname>Nguyen</surname><given-names>DQ</given-names> </name><name name-style="western"><surname>Dohopolski</surname><given-names>M</given-names> </name><name name-style="western"><surname>Rousseau</surname><given-names>JF</given-names> </name><name name-style="western"><surname>Peterson</surname><given-names>ED</given-names> </name><name name-style="western"><surname>Navar</surname><given-names>AM</given-names> </name></person-group><article-title>Accuracy of large language models to identify stroke subtypes within unstructured electronic health record data</article-title><source>Stroke</source><year>2025</year><month>10</month><volume>56</volume><issue>10</issue><fpage>2966</fpage><lpage>2975</lpage><pub-id pub-id-type="doi">10.1161/STROKEAHA.125.051993</pub-id><pub-id pub-id-type="medline">40709446</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Van Veen</surname><given-names>D</given-names> </name><name name-style="western"><surname>Van Uden</surname><given-names>C</given-names> </name><name name-style="western"><surname>Blankemeier</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Adapted large language models can outperform medical experts in clinical text summarization</article-title><source>Nat Med</source><year>2024</year><month>04</month><volume>30</volume><issue>4</issue><fpage>1134</fpage><lpage>1142</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-02855-5</pub-id><pub-id pub-id-type="medline">38413730</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>JT</given-names> </name><name name-style="western"><surname>Li</surname><given-names>VC</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>JJ</given-names> </name><etal/></person-group><article-title>Evaluation of performance of generative large language models for stroke care</article-title><source>NPJ Digit Med</source><year>2025</year><month>07</month><day>29</day><volume>8</volume><issue>1</issue><fpage>481</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01830-9</pub-id><pub-id pub-id-type="medline">40730644</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Beutel</surname><given-names>G</given-names> </name><name name-style="western"><surname>Geerits</surname><given-names>E</given-names> </name><name name-style="western"><surname>Kielstein</surname><given-names>JT</given-names> </name></person-group><article-title>Artificial hallucination: GPT on LSD?</article-title><source>Crit Care</source><year>2023</year><month>04</month><day>18</day><volume>27</volume><issue>1</issue><fpage>148</fpage><pub-id pub-id-type="doi">10.1186/s13054-023-04425-6</pub-id><pub-id pub-id-type="medline">37072798</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Farquhar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kossen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kuhn</surname><given-names>L</given-names> </name><name name-style="western"><surname>Gal</surname><given-names>Y</given-names> </name></person-group><article-title>Detecting hallucinations in large language models using semantic entropy</article-title><source>Nature</source><year>2024</year><month>06</month><volume>630</volume><issue>8017</issue><fpage>625</fpage><lpage>630</lpage><pub-id pub-id-type="doi">10.1038/s41586-024-07421-0</pub-id><pub-id pub-id-type="medline">38898292</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Williams</surname><given-names>CY</given-names> </name><name name-style="western"><surname>Bains</surname><given-names>J</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Evaluating large language models for drafting emergency department encounter summaries</article-title><source>PLOS Digit Health</source><year>2025</year><month>06</month><volume>4</volume><issue>6</issue><fpage>e0000899</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000899</pub-id><pub-id pub-id-type="medline">40526634</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="web"><article-title>Enhancing clinical decision-making in acute ischemic stroke with a hybrid-reasoning framework based on large language models</article-title><source>Anonymous GitHub</source><access-date>2026-07-27</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://anonymous.4open.science/r/HR-LLM-Stroke/README.md">https://anonymous.4open.science/r/HR-LLM-Stroke/README.md</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Supplementary results, including study cohort flow and additional analyses.</p><media xlink:href="jmir_v28i1e96304_app1.docx" xlink:title="DOCX File, 1865 KB"/></supplementary-material></app-group></back></article>