<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e91257</article-id><article-id pub-id-type="doi">10.2196/91257</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>A Multidisciplinary Team&#x2013;Based Large Language Model Framework for Predicting Postoperative Neurological Complications in Acute Type A Aortic Dissection: Model Development and Validation Study</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Li</surname><given-names>Jili</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Zhang</surname><given-names>Julin</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Tao</surname><given-names>Xingrui</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chen</surname><given-names>Yaoye</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Xun</surname><given-names>Siqi</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Qin</surname><given-names>Guangshuo</given-names></name><degrees>BDS</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Diao</surname><given-names>Kaiyue</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Qin</surname><given-names>Chaoyi</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Thoracic Surgery and Institute of Thoracic Oncology, West China Hospital, Sichuan University</institution><addr-line>Chengdu</addr-line><addr-line>Sichuan</addr-line><country>China</country></aff><aff id="aff2"><institution>Department of Cardiovascular Surgery and Cardiovascular Surgery Research Laboratory, West China Hospital, Sichuan University</institution><addr-line>No. 37, Guoxue Alley</addr-line><addr-line>Chengdu</addr-line><addr-line>Sichuan</addr-line><country>China</country></aff><aff id="aff3"><institution>West China School of Medicine, West China Hospital, Sichuan University</institution><addr-line>Chengdu</addr-line><addr-line>Sichuan</addr-line><country>China</country></aff><aff id="aff4"><institution>School of Computer Science, Shanghai Jiao Tong University</institution><addr-line>Shanghai</addr-line><country>China</country></aff><aff id="aff5"><institution>Department of Radiology, West China Hospital, Sichuan University</institution><addr-line>Chengdu</addr-line><addr-line>Sichuan</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Li</surname><given-names>Chenyu</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Al-Agil</surname><given-names>Mohammad</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Chaoyi Qin, MD, Department of Cardiovascular Surgery and Cardiovascular Surgery Research Laboratory, West China Hospital, Sichuan University, No. 37, Guoxue Alley, Chengdu, Sichuan, 610041, China, 86 028 85422897; <email>qinchaoyi@wchscu.edu.cn</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>18</day><month>8</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e91257</elocation-id><history><date date-type="received"><day>12</day><month>01</month><year>2026</year></date><date date-type="rev-recd"><day>21</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>23</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Jili Li, Julin Zhang, Xingrui Tao, Yaoye Chen, Siqi Xun, Guangshuo Qin, Kaiyue Diao, Chaoyi Qin. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 18.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e91257"/><abstract><sec><title>Background</title><p>Postoperative neurological complications (PNCs) after acute type A aortic dissection (ATAAD) surgery are clinically emergent and require multidimensional perioperative risk assessment. Large language models (LLMs) have shown potential in clinical prediction, but the incremental value of structured multiagent collaboration remains unclear.</p></sec><sec><title>Objective</title><p>This study aimed to develop and validate a multidisciplinary team (MDT)&#x2013;based LLM framework for predicting PNC after ATAAD surgery and to compare its performance with that of traditional machine learning (ML) models and single-agent LLM settings.</p></sec><sec sec-type="methods"><title>Methods</title><p>A retrospective cohort from January 2020 to June 2024 (N=763) was randomly divided into a training set (n=533) and an internal validation set (n=230). A prospective cohort from July 2024 to June 2025 (n=120) was used for prospective validation. The outcome was PNC, defined as stroke, cerebral hemorrhage, paraplegia, or coma. Population-level in-context learning used outcome-stratified summary statistics from the training cohort, including predictor distributions in patients with and without PNC and between-group <italic>P</italic> values. Two LLMs (DeepSeek-V3 and ChatGPT [GPT-5]) were evaluated under 4 settings: no MDT without in-context learning, MDT without in-context learning, no MDT with in-context learning, and MDT with in-context learning. Model performance was assessed using the area under the receiver operating characteristic curve (AUC), sensitivity, specificity, accuracy, <italic>F</italic><sub>1</sub>-score, and Brier score.</p></sec><sec sec-type="results"><title>Results</title><p>PNC occurred in 13.0% (99/763) of patients in the retrospective cohort and 15.8% (19/120) in the prospective cohort. Among the ML models, the random forest achieved the highest AUC in the internal validation set (AUC 0.7857, 95% CI 0.6945&#x2010;0.8768). Among the LLM configurations, GPT-5 with MDT and in-context learning achieved the highest AUC (AUC 0.8419, 95% CI 0.7398&#x2010;0.9440), with a sensitivity of 80.00%, specificity of 87.00%, and Brier score of 0.0876. Although this configuration showed a significantly higher AUC than the fully unaided GPT-5 baseline (<italic>P</italic>=.006), this improvement reflected the combined effect of MDT and in-context learning, as adding the MDT framework within matched settings did not significantly improve AUC over corresponding single-agent settings in either cohort. The AUC of GPT-5 with MDT and in-context learning was also not significantly higher than that of the random forest model (<italic>P</italic>=.18). Word frequency analysis showed that the generated rationales following in-context learning more frequently mentioned variables that were statistically significant in the provided context.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>The GPT-5 configuration combining MDT-style collaboration with population-level in-context learning showed promising discrimination for PNC prediction after ATAAD surgery. However, the MDT layer did not produce a statistically significant incremental improvement in AUC over the corresponding single-agent settings, and its predictive contribution remains to be confirmed. The framework generated structured, role-specific rationales, supporting its further evaluation as a proof-of-concept approach for postoperative risk stratification. Larger multicenter studies are required before routine clinical implementation.</p></sec></abstract><kwd-group><kwd>acute type A aortic dissection</kwd><kwd>postoperative neurological complications</kwd><kwd>large language model</kwd><kwd>multidisciplinary team</kwd><kwd>in-context learning</kwd><kwd>machine learning</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Prediction plays a critical role in the health care domain, with common application scenarios including assessing disease mortality [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>], evaluating drug suitability [<xref ref-type="bibr" rid="ref4">4</xref>-<xref ref-type="bibr" rid="ref6">6</xref>], and determining patient discharge readiness [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. Traditional machine learning (ML) methods have been widely used in clinical prediction tasks and have demonstrated good performance [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. With the advancement of large language models (LLMs), various models, such as ChatGPT and the DeepSeek series, have emerged. These models perform well in text understanding and generation [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref12">12</xref>], and their application in clinical prediction tasks is increasing [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref14">14</xref>]. Prior studies have also evaluated LLMs using standardized clinical vignettes [<xref ref-type="bibr" rid="ref15">15</xref>], physician-authored case summaries [<xref ref-type="bibr" rid="ref16">16</xref>], or structured electronic health record information converted into narrative prompts [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>]. However, because LLMs are built upon diverse, general-purpose databases not specifically designed for the clinical domain, they possess universality and broad-spectrum applicability that can limit their professional use in clinical applications. A study indicated that LLMs perform less effectively than locally trained traditional ML models in clinical prediction tasks [<xref ref-type="bibr" rid="ref19">19</xref>] and that they exhibit limitations when handling complex cases [<xref ref-type="bibr" rid="ref20">20</xref>]. Furthermore, the real-world &#x201C;out-of-the-box&#x201D; performance of LLMs in actual health care settings remains unclear, and strategies for enhancing their capabilities through precise in-context learning require further exploration.</p><p>Acute type A aortic dissection (ATAAD) is a life-threatening cardiovascular emergency [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref22">22</xref>]. Surgical intervention remains the primary treatment for patients with ATAAD. While surgical mortality for ATAAD has significantly decreased with advancements in surgical techniques, postoperative in-hospital mortality remains relatively high, reported to be up to 20% [<xref ref-type="bibr" rid="ref23">23</xref>]. The occurrence of postoperative neurological complications (PNCs) significantly impacts patient outcomes, leading to postoperative cognitive dysfunction and an increased risk of in-hospital mortality. Therefore, the timely prediction, identification, and intervention for PNC represent a crucial and valuable clinical task for improving patient outcomes. The probability of PNC is influenced by multiple preoperative and intraoperative factors, such as preoperative altered mental status [<xref ref-type="bibr" rid="ref24">24</xref>], intraoperative peak lactate level [<xref ref-type="bibr" rid="ref25">25</xref>], cannulation site [<xref ref-type="bibr" rid="ref26">26</xref>], and cerebral perfusion strategies [<xref ref-type="bibr" rid="ref27">27</xref>]. This typically necessitates collaborative diagnosis and decision-making by a multidisciplinary medical team comprising specialists in cardiac surgery, anesthesiology, and neurology.</p><p>Currently, manually organized multidisciplinary team (MDT) medical consultations consume substantial human, material, and financial resources [<xref ref-type="bibr" rid="ref28">28</xref>]. Moreover, ATAAD presents abruptly and progresses rapidly. Without timely intervention, the mortality rate reaches as high as 50% within 48 hours, increasing by 1% to 2% per hour after symptom onset. Given the critical time constraints for patients, there is often insufficient time for conventional MDT medical consultations, thus creating an urgent need for a more convenient and efficient collaborative diagnostic approach. In clinical applications, different agents can be designed to organize structured, role-specific discussions, inspired by MDT collaboration, for complex medical issues [<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref30">30</xref>], although such systems are not equivalent to real MDT consultations that also involve bedside examination, imaging review, longitudinal patient knowledge, and nuanced clinical judgment [<xref ref-type="bibr" rid="ref31">31</xref>]. The multiagent framework represents an innovative approach that significantly enhances the capabilities of LLMs to handle complex tasks [<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>] by enabling multiple agents to engage in discussions around the same problem and ultimately reach a consensus on the output. Simultaneously, the resulting conversation provides a traceable record of role-specific reasoning, allowing health care professionals to review the model&#x2019;s deliberation pathway and increasing its credibility in clinical applications.</p><p>This study aims to develop an MDT framework for clinical prediction tasks. By using an in-context learning method, LLMs can learn from typical clinical features and reduce the limitations imposed by the length of the context text on the number of training cases. In particular, we systematically compare the predictive performance of traditional ML, single agent&#x2013;based LLMs, and MDT framework&#x2013;based LLMs in assessing the risk of PNC following ATAAD surgery.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Ethical Considerations</title><p>This study was conducted in accordance with the Declaration of Helsinki and was approved by the Institutional Review Board of West China Hospital, Sichuan University (2023-1999). Given the retrospective nature of the initial cohort development and the anonymization of data, the requirement for informed consent was waived. However, written informed consent was obtained from all participants enrolled in the prospective validation cohort. To ensure privacy and confidentiality, all patient data were deidentified prior to analysis. In accordance with our data use agreement, all prompts transmitted to the LLM application programming interfaces were rigorously monitored to ensure that no protected health information was included.</p></sec><sec id="s2-2"><title>Study Population</title><p>This study consisted of 2 independent, nonoverlapping cohorts of patients diagnosed with ATAAD at West China Hospital, Sichuan University. In the retrospective cohort derived from the electronic medical record system, after excluding those with chronic aortic dissection (n=139) from an initial pool of 902 patients, a total of 763 patients diagnosed with ATAAD were included. The retrospective cohort (January 2020 to June 2024, n=763) was used for model development and internal validation, while the prospective cohort (July 2024 to June 2025, n=120) was used for prospective validation. The predicted outcome was the occurrence of PNC, defined as a composite of stroke, cerebral hemorrhage, paraplegia, or coma. Stroke was defined as a new focal or global neurological deficit attributable to cerebral ischemia, with compatible cranial computed tomography or magnetic resonance imaging findings when imaging was available or clinically indicated. Cerebral hemorrhage was defined as postoperative intracranial hemorrhage confirmed by cranial computed tomography or magnetic resonance imaging. Paraplegia was defined as a new postoperative bilateral lower-extremity motor deficit consistent with spinal cord ischemia and documented by neurological examination during hospitalization. Coma was defined as unarousable unconsciousness without purposeful response to verbal or painful stimulation lasting more than 48 hours after surgery after excluding residual anesthesia or sedation. Predictors encompassed demographics, comorbidities, laboratory profiles, imaging findings, and operative details. The intended clinical use of this framework was postoperative risk assessment and early prognostic stratification for PNC after ATAAD surgery. Therefore, the predictor set was defined to include complete perioperative information, including intraoperative variables, to reflect the way postoperative neurological risk is commonly assessed in routine clinical practice.</p><p>The development, analysis, and reporting of this clinical prediction model study were performed in accordance with the TRIPOD+AI (Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis + AI) statement (<xref ref-type="supplementary-material" rid="app2">Checklist 1</xref>), which provides updated guidance for reporting prediction models developed using regression or machine learning methods [<xref ref-type="bibr" rid="ref34">34</xref>]. For LLM-based prediction, the prespecified variables extracted from the electronic medical records were further transformed into standardized clinical vignettes. A uniform vignette template was manually designed based on the study variables, and Python scripts were used to batch-convert the structured variables into narrative case descriptions for model input. This process was completed by 1 author (SX), who had access only to the predictor variables and was blinded to the prediction outcomes, thereby reducing the risk of information bias.</p></sec><sec id="s2-3"><title>ML Models and Explainability</title><p>Four machine learning models (random forest, extreme gradient boosting, Gaussian Naive Bayes, and logistic regression) were developed to predict PNC. For the primary analysis, the ML models were trained and evaluated using the same set of predictor variables as those provided to the LLMs, including demographics, comorbidities, laboratory profiles, imaging findings, and operative variables. As a sensitivity analysis, we additionally repeated the ML modeling using the 15 predictors selected by least absolute shrinkage and selection operator (LASSO) regression, which was performed strictly within the training set of 533 patients to reduce model dimensionality and mitigate overfitting. LASSO regression was performed using the glmnet R package (version 4.1-10; R Foundation for Statistical Computing), with categorical predictors one-hot encoded and predictors standardized before model fitting. The penalty parameter &#x03BB; was selected by 5-fold cross-validation using the value that minimized cross-validated binomial deviance. Predictors with nonzero coefficients at the optimal &#x03BB; were retained for the sensitivity analysis. The corresponding results are presented in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. The random forest model, which demonstrated the highest predictive performance among ML models, was subsequently analyzed to quantify the global importance of each input predictor using Shapley Additive Explanations (SHAP) [<xref ref-type="bibr" rid="ref35">35</xref>].</p></sec><sec id="s2-4"><title>LLM In-Context Learning</title><p>DeepSeek-V3 and ChatGPT (GPT-5), which are increasingly being evaluated for clinical applications, were selected as the base models. The temperature for DeepSeek-V3 was set to 0 to reduce output stochasticity and improve reproducibility. For GPT-5, which does not support a user-controlled temperature parameter according to the official document, we used the default model configuration. To assess the reproducibility of GPT-5 predictions under the default model configuration, we conducted a repeated-inference analysis on a stratified random sample of 50 patients from the retrospective cohort, selected according to the observed outcome distribution. For each patient, inference was independently repeated 5 times under identical prompting conditions across all 4 GPT-5 application settings, using all clinical variables. Within-patient variability of predicted probabilities was quantified using the SD, range, and IQR.</p><p>In addition, we adopted an in-context learning strategy to prime the LLMs using aggregated, population-level summary statistics. Rather than updating model parameters or fine-tuning the models, these statistics were injected into the system prompt at inference time as structured contextual information. This approach avoided the computational cost of fine-tuning while providing the models with essential background knowledge. The background knowledge, formatted in JSON, included distributions of the predictors across the entire training cohort, the PNC group, and the non-PNC group, along with <italic>P</italic> values from intergroup comparisons (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). This summary enabled the LLMs to recognize key risk factors and population characteristics associated with PNC. By leveraging these statistical summaries, the LLMs established a foundational understanding of the clinical dataset, supporting subsequent reasoning for individual patient predictions and ensuring an equitable comparison with traditional ML models.</p><p>For the primary analysis, both the cohort-level background knowledge and the patient-specific prompts were constructed using exactly the same full set of structured clinical variables available in the original dataset, as that used for the ML models, thereby ensuring a fair comparison. As a sensitivity analysis, the LLM-based predictions were repeated using the same 15 predictors selected by LASSO regression, and the background knowledge also contained only the selected variables (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p></sec><sec id="s2-5"><title>Multiagent Collaborative Framework</title><p>The same models were selected for our multiagent conversation framework developed using AutoGen (Microsoft) [<xref ref-type="bibr" rid="ref36">36</xref>]. As shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>, this framework simulates MDT consultations for predicting PNC. The framework comprises 3 specialized agents (cardiovascular surgeon, anesthesiologist, and neurologist) and 1 supervisor agent. Each specialist agent analyzes the case from a distinct perspective: the cardiovascular surgeon focuses on surgical factors and anatomy, the anesthesiologist evaluates intraoperative management and hemodynamics, and the neurologist assesses neurological status and cerebrovascular conditions. For the no MDT baseline, predictions were generated by the single cardiovascular surgeon agent because patients undergoing ATAAD surgery are primarily managed in the cardiovascular surgery ward after surgery, and early postoperative risk assessment is commonly initiated by cardiovascular surgeons in routine clinical practice. The same single-agent role and prompt schema were used for both DeepSeek-V3 and GPT-5 under the no MDT without in-context learning and no MDT with in-context learning settings. The complete prompts for these 2 no MDT settings are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Schematic of the multiagent collaborative framework for predicting postoperative neurological complications. The multidisciplinary team (MDT) framework, built with AutoGen, comprises 3 specialist agents (cardiovascular surgeon, anesthesiologist, and neurologist) and a supervisor agent, simulating an MDT consultation to reach a consensus-based prediction.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e91257_fig01.png"/></fig><p>For each patient case, the conversation was initialized with an empty message history, so no conversational state was carried over between cases. Patient-specific narrative prompts were generated from the completed dataset after imputation, and imputed values were not explicitly labeled or distinguished from observed clinical measurements in the prompts. Within each case-level session, the dialog state was maintained through the accumulated group-chat history. The dialog followed a structured framework, initiated by the supervisor, who presented the patient&#x2019;s information and guided each specialist agent to provide a unique probability and rationale. After each discussion cycle, the supervisor coordinated exchanges and monitored for consensus, defined as a probability range of &#x2264;0.1. Once consensus was reached, or after a maximum of 17 rounds, the dialog terminated. The supervisor then produced the average probability and a summary rationale, integrating complementary expertise through a structured MDT-style deliberation. The complete conversation history and the final supervisor summary were saved for subsequent analysis. To characterize the content emphasized in the multiagent outputs, we conducted a word-frequency analysis of the summary rationales generated by the supervisor.</p></sec><sec id="s2-6"><title>Statistical Analysis</title><p>We handled missing data using multiple imputation by chained equations (MICE), a well-established approach for addressing uncertainty in imputed values [<xref ref-type="bibr" rid="ref37">37</xref>]. For the retrospective cohort, missing continuous predictors were imputed using MICE with 5 imputed datasets and 10 iterations, and the averaged values were then used to construct the completed dataset. For the prospective validation cohort, we enrolled 120 consecutive eligible patients for whom the prespecified predictor variables were fully available, and therefore no imputation was performed. In addition, to ensure a fair comparison, traditional ML models and LLM-based models were evaluated using the same set of predictor variables on the completed dataset after imputation. Continuous variables were assessed for normality using the Shapiro-Wilk test. Nonnormally distributed variables were summarized as medians with IQRs and compared using the Mann-Whitney <italic>U</italic> test, whereas normally distributed variables were presented as means with SDs and compared using the 2-tailed <italic>t</italic> test. Categorical variables were expressed as numbers and proportions, with group comparisons performed using the chi-square test or Fisher exact test, as appropriate.</p><p>The performance of both ML models and LLMs was evaluated using the area under the receiver operating characteristic curve (AUC), sensitivity, specificity, positive predictive value (PPV), negative predictive value (NPV), accuracy, <italic>F</italic><sub>1</sub>-score, and Brier score. To achieve optimal classification performance, the prediction threshold for each model was selected by maximizing the Youden index. The 95% CIs for AUCs were calculated using the DeLong method, whereas those for sensitivity, specificity, PPV, NPV, accuracy, <italic>F</italic><sub>1</sub>-score, and Brier score were estimated using 1000-iteration nonparametric bootstrap resampling, with the Youden index threshold reestimated in each bootstrap sample for threshold-dependent metrics. To compare discriminative performance between models, differences in AUC were assessed using the DeLong test. All <italic>P</italic> values from model-comparison tests were reported as unadjusted exploratory results, and the study was not powered for confirmatory multimodel hypothesis testing across all LLM configurations, cohorts, and model classes. Decision curve analysis and calibration plots were performed to further assess the clinical utility and probability calibration of the LLM models. All statistical analyses were performed using Python (version 3.13.7; Python Software Foundation) and R software (version 4.5.0).</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Patient Characteristics</title><p>The study comprised a retrospective cohort of 763 patients and a prospective validation cohort of 120 patients with ATAAD. We randomly allocated the retrospective cohort to a training set (n=533, 70%) and an internal validation set (n=230, 30%). The overall incidence of PNC in the retrospective cohort was 13.0% (99/763). Detailed comparisons of demographic, clinical, and operative characteristics between patients with and without PNC in the retrospective cohort are provided in Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. In the prospective validation cohort, PNC occurred in 19 of 120 (15.8%) patients. The baseline characteristics of the retrospective and prospective validation cohorts are summarized in Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. The details of the study design are presented in <xref ref-type="fig" rid="figure2">Figure 2</xref>.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Study workflow for developing and validating machine learning (ML) and large language model (LLM) methods to predict postoperative neurological complications (PNC). The retrospective cohort was randomly divided into a training set and an internal validation set (7:3). After data preprocessing and variable selection, 4 ML models and single-agent or multiagent LLM configurations, with or without population-level in-context learning, were evaluated. Model performance was assessed using AUC, sensitivity, specificity, accuracy, <italic>F</italic><sub>1</sub>-score, and Brier score, and interpretability was examined using the Shapley Additive Explanations (SHAP) algorithm and word frequency analysis. ATAAD: acute type A aortic dissection; AUC: area under the receiver operating characteristic curve; LASSO: least absolute shrinkage and selection operator.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e91257_fig02.png"/></fig></sec><sec id="s3-2"><title>ML Model and Interpretation</title><p>We developed 4 ML models using all prespecified candidate predictors, exactly matching the variables provided to the LLMs in the primary analysis. As shown in <xref ref-type="table" rid="table1">Table 1</xref> and <xref ref-type="fig" rid="figure3">Figure 3</xref>, the random forest achieved the highest AUC on the internal validation set among the 4 ML models (AUC 0.7857, 95% CI 0.6945&#x2010;0.8768), whereas logistic regression had the poorest predictive performance (AUC 0.7292, 95% CI 0.6330&#x2010;0.8254).</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Performance of each machine learning model for prediction on the internal validation set using all predictor variables.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">AUC<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> (95% CI)</td><td align="left" valign="bottom">Sensitivity (95% CI)</td><td align="left" valign="bottom">Specificity (95% CI)</td><td align="left" valign="bottom">PPV<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> (95% CI)</td><td align="left" valign="bottom">NPV<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup> (95% CI)</td><td align="left" valign="bottom">Accuracy (95% CI)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="left" valign="bottom">Brier score (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">Random Forest</td><td align="left" valign="top">0.7857 (0.6945-0.8768)</td><td align="left" valign="top">0.6333 (0.5217-0.9677)</td><td align="left" valign="top">0.8550 (0.4368-0.9114)</td><td align="left" valign="top">0.3958 (0.1806-0.5557)</td><td align="left" valign="top">0.9396 (0.9124-0.9916)</td><td align="left" valign="top">0.8261 (0.4955-0.8784)</td><td align="left" valign="top">0.4872 (0.3043-0.6119)</td><td align="left" valign="top">0.1020 (0.0727-0.1314)</td></tr><tr><td align="left" valign="top">XGBoost</td><td align="left" valign="top">0.7450 (0.6419-0.8481)</td><td align="left" valign="top">0.8333 (0.4705-0.9565)</td><td align="left" valign="top">0.6200 (0.5604-0.9415)</td><td align="left" valign="top">0.2475 (0.1731-0.6088)</td><td align="left" valign="top">0.9612 (0.9179-0.9919)</td><td align="left" valign="top">0.6478 (0.5957-0.8871)</td><td align="left" valign="top">0.3817 (0.2836-0.5582)</td><td align="left" valign="top">0.1016 (0.0717-0.1328)</td></tr><tr><td align="left" valign="top">Naive Bayes</td><td align="left" valign="top">0.7634 (0.6678-0.8591)</td><td align="left" valign="top">0.8333 (0.6295-0.9600)</td><td align="left" valign="top">0.6850 (0.6231-0.8707)</td><td align="left" valign="top">0.2841 (0.1932-0.4194)</td><td align="left" valign="top">0.9648 (0.9296-0.9926)</td><td align="left" valign="top">0.7043 (0.6478-0.8435)</td><td align="left" valign="top">0.4237 (0.3106-0.5393)</td><td align="left" valign="top">0.2761 (0.2288-0.3313)</td></tr><tr><td align="left" valign="top">Logistic regression</td><td align="left" valign="top">0.7292 (0.6330-0.8254)</td><td align="left" valign="top">0.7333 (0.5454-0.9714)</td><td align="left" valign="top">0.6700 (0.4349-0.8443)</td><td align="left" valign="top">0.2500 (0.1648-0.3969)</td><td align="left" valign="top">0.9437 (0.9130-0.9921)</td><td align="left" valign="top">0.6783 (0.4957-0.8174)</td><td align="left" valign="top">0.3729 (0.2745-0.5037)</td><td align="left" valign="top">0.1141 (0.0839-0.1446)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>AUC: area under the receiver operating characteristic curve.</p></fn><fn id="table1fn2"><p><sup>b</sup>PPV: positive predictive value.</p></fn><fn id="table1fn3"><p><sup>c</sup>NPV: negative predictive value.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Receiver operating characteristic (ROC) curves of the 4 machine learning models.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e91257_fig03.png"/></fig><p>To provide visual interpretations of the prediction outcome using the internal validation set, we applied the SHAP algorithm to explain the random forest model and displayed the top 10 predictors ranked by importance in <xref ref-type="fig" rid="figure4">Figure 4A</xref>. The intraoperative peak lactate was the most important feature for predicting postoperative PNC, followed closely by the preoperative myoglobin level. The association between the SHAP values of each feature and their effects on the model prediction is presented in <xref ref-type="fig" rid="figure4">Figure 4B</xref>. The color of the dots indicates whether the feature value is high (red) or low (blue) for each observation.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Interpretation of the random forest model using the Shapley Additive Explanations (SHAP) method. (A) Global feature importance is ranked by mean absolute SHAP values. (B) The SHAP summary plot shows the direction and magnitude of each feature&#x2019;s contribution to the predicted risk of postoperative neurological complications.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e91257_fig04.png"/></fig></sec><sec id="s3-3"><title>LLM Prediction and Multiagent Evaluation</title><p>Repeated inference demonstrated limited within-patient variability across all 4 GPT-5 application settings (Figure S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The mean within-patient SD ranged from 0.015 to 0.023, the mean range from 0.036 to 0.054, and the mean IQR from 0.018 to 0.028 (Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The GPT-5+MDT+context setting showed relatively low variability, with a mean SD of 0.015 and a mean range of 0.036, supporting acceptable empirical reproducibility under fixed prompting conditions. Based on the reproducibility of LLMs, we evaluated the predictive performance of DeepSeek-V3 and GPT-5 under 4 application settings in the internal validation set. As summarized in <xref ref-type="table" rid="table2">Table 2</xref>, configurations incorporating in-context learning had higher observed AUCs than the corresponding settings without in-context learning. In the internal validation set, GPT-5 achieved a higher AUC than DeepSeek-V3 in each matched configuration. The highest AUC was observed for GPT-5 with both MDT and in-context learning (AUC 0.8419, 95% CI 0.7398&#x2010;0.9440), with a sensitivity of 80.00%, specificity of 87.00%, accuracy of 86.09%, and <italic>F</italic><sub>1</sub>-score of 0.60. Compared with the fully unaided baseline without MDT or in-context learning, the MDT+context configuration showed a significantly higher AUC for both GPT-5 (<italic>P</italic>=.006) and DeepSeek-V3 (<italic>P</italic>=.01). However, these contrasts reflect the combined addition of MDT and in-context learning and do not isolate the contribution of the MDT layer. In matched comparisons, adding MDT yielded numerically higher AUCs than the corresponding settings without MDT, both without in-context learning (GPT-5: 0.7840 vs 0.7031, <italic>P</italic>=.09; DeepSeek-V3: 0.7458 vs 0.6533, <italic>P</italic>=.22) and with in-context learning (GPT-5: 0.8419 vs 0.8126, <italic>P</italic>=.35; DeepSeek-V3: 0.8003 vs 0.7719, <italic>P</italic>=.55), but none of these differences reached statistical significance. The AUC of GPT-5 with MDT and in-context learning was also not significantly higher than that of the random forest model (P=.18).</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Performance of each large language model for prediction on the internal validation set using all predictor variables.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">AUC<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> (95% CI)</td><td align="left" valign="bottom">Sensitivity (95% CI)</td><td align="left" valign="bottom">Specificity (95% CI)</td><td align="left" valign="bottom">PPV<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup> (95% CI)</td><td align="left" valign="bottom">NPV<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup> (95% CI)</td><td align="left" valign="bottom">Accuracy (95% CI)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="left" valign="bottom">Brier score (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="9">DeepSeek-V3</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No MDT<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup>+no context</td><td align="left" valign="top">0.6533 (0.5515-0.7552)</td><td align="left" valign="top">0.7000 (0.2500-0.9630)</td><td align="left" valign="top">0.5650 (0.2249-0.9511)</td><td align="left" valign="top">0.1944 (0.1290-0.4545)</td><td align="left" valign="top">0.9262 (0.8823-0.9832)</td><td align="left" valign="top">0.5826 (0.3216-0.8696)</td><td align="left" valign="top">0.3043 (0.2182-0.4139)</td><td align="left" valign="top">0.1453 (0.1275-0.1623)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MDT+no context</td><td align="left" valign="top">0.7458 (0.6357-0.8560)</td><td align="left" valign="top">0.6667 (0.4346-0.8400)</td><td align="left" valign="top">0.8000 (0.7526-0.9476)</td><td align="left" valign="top">0.3333 (0.2499-0.6002)</td><td align="left" valign="top">0.9412 (0.9000-0.9763)</td><td align="left" valign="top">0.7826 (0.7391-0.9043)</td><td align="left" valign="top">0.4444 (0.3376-0.6217)</td><td align="left" valign="top">0.1064 (0.0834-0.1304)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No MDT+context</td><td align="left" valign="top">0.7719 (0.6796-0.8642)</td><td align="left" valign="top">0.6667 (0.5556-0.9375)</td><td align="left" valign="top">0.8150 (0.5937-0.8756)</td><td align="left" valign="top">0.3509 (0.1935-0.4728)</td><td align="left" valign="top">0.9422 (0.9200-0.9862)</td><td align="left" valign="top">0.7957 (0.6304-0.8479)</td><td align="left" valign="top">0.4598 (0.3103-0.5834)</td><td align="left" valign="top">0.1432 (0.1194-0.1676)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MDT+context</td><td align="left" valign="top">0.8003 (0.6944-0.9062)</td><td align="left" valign="top">0.7000 (0.5312-0.8889)</td><td align="left" valign="top">0.8650 (0.7339-0.9453)</td><td align="left" valign="top">0.4375 (0.2698-0.6667)</td><td align="left" valign="top">0.9505 (0.9176-0.9830)</td><td align="left" valign="top">0.8435 (0.7435-0.9174)</td><td align="left" valign="top">0.5385 (0.3859-0.6977)</td><td align="left" valign="top">0.1091 (0.0909-0.1285)</td></tr><tr><td align="left" valign="top" colspan="9">GPT-5</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No MDT+no context</td><td align="left" valign="top">0.7031 (0.6001-0.8061)</td><td align="left" valign="top">0.7667 (0.3793-0.9355)</td><td align="left" valign="top">0.5950 (0.4950-0.9344)</td><td align="left" valign="top">0.2212 (0.1509-0.4800)</td><td align="left" valign="top">0.9444 (0.8985-0.9835)</td><td align="left" valign="top">0.6174 (0.5348-0.8696)</td><td align="left" valign="top">0.3433 (0.2520-0.4878)</td><td align="left" valign="top">0.1246 (0.1045-0.1443)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MDT+no context</td><td align="left" valign="top">0.7840 (0.6820-0.8860)</td><td align="left" valign="top">0.7000 (0.5416-0.9394)</td><td align="left" valign="top">0.8150 (0.5634-0.9015)</td><td align="left" valign="top">0.3621 (0.2066-0.5385)</td><td align="left" valign="top">0.9477 (0.9167-0.9882)</td><td align="left" valign="top">0.8000 (0.6087-0.8783)</td><td align="left" valign="top">0.4773 (0.3333-0.6316)</td><td align="left" valign="top">0.1151 (0.0982-0.1327)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No MDT+context</td><td align="left" valign="top">0.8126 (0.7216-0.9036)</td><td align="left" valign="top">0.7000 (0.5185-0.8948)</td><td align="left" valign="top">0.8350 (0.6250-0.9436)</td><td align="left" valign="top">0.3889 (0.2208-0.6071)</td><td align="left" valign="top">0.9489 (0.9171-0.9808)</td><td align="left" valign="top">0.8174 (0.6434-0.9043)</td><td align="left" valign="top">0.5000 (0.3422-0.6429)</td><td align="left" valign="top">0.1089 (0.0928-0.1260)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>MDT+context</td><td align="left" valign="top">0.8419 (0.7398-0.9440)</td><td align="left" valign="top">0.8000 (0.6296-0.9310)</td><td align="left" valign="top">0.8700 (0.8308-0.9801)</td><td align="left" valign="top">0.4800 (0.3599-0.8333)</td><td align="left" valign="top">0.9667 (0.9375-0.9892)</td><td align="left" valign="top">0.8609 (0.8217-0.9478)</td><td align="left" valign="top">0.6000 (0.4839-, 0.7857)</td><td align="left" valign="top">0.0876 (0.0701-0.1069)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>AUC: area under the receiver operating characteristic curve.</p></fn><fn id="table2fn2"><p><sup>b</sup>PPV: positive predictive value.</p></fn><fn id="table2fn3"><p><sup>c</sup>NPV: negative predictive value.</p></fn><fn id="table2fn4"><p><sup>d</sup>MDT: multidisciplinary team.</p></fn></table-wrap-foot></table-wrap><p>Decision curve analysis was performed to examine the net clinical benefit of the LLM-based models, with interpretation focused on the clinically relevant threshold probability range of 0.10 to 0.30 for postoperative early warning and risk stratification. Within this range, the GPT-5+MDT+context configuration showed the most favorable overall net benefit among the GPT-5 settings, particularly around thresholds of 0.20 to 0.30 (Figure S2A in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). In contrast, the DeepSeek-V3 models showed less stable decision curves, and the MDT+context configuration demonstrated a positive net benefit only within part of the clinically relevant range, without consistently outperforming the no MDT+context setting (Figure S2C in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Calibration was also assessed using Brier scores and calibration plots. Among the GPT-5-based configurations, the in-context learning settings showed lower Brier scores than the no-context settings, with MDT+context yielding the lowest Brier score (0.0876) and showing closer agreement between predicted probabilities and observed event rates (Figure S2B in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). For DeepSeek-V3, the MDT+context configuration also yielded a relatively low Brier score (0.1091), although the calibration curves remained more variable overall (Figure S2D in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p><p>In addition, the dynamic, multiround consultation process among the specialized agents was graphically depicted in a chat-based format in <xref ref-type="fig" rid="figure5">Figure 5</xref>. To elucidate the distinct reasoning processes and outputs under each configuration, representative examples for all experimental conditions were provided in the Supplementary Examples in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Supplementary Example A illustrates the baseline prediction generated by a single-agent LLM without in-context learning. Supplementary Example B demonstrates the output from the LLM model after in-context learning with population-level statistical summaries. Supplementary Example C provides a transcript of the multiagent discussion conducted without prior in-context learning. Supplementary Example D presents a full 17-round MDT dialog augmented with in-context learning, demonstrating the integration of specialized expertise and background data in evaluating complex risk factors, even in the absence of full consensus. These examples illustrate the collaborative dynamics facilitated by the proposed framework. Nonconsensus events after a maximum of 17 rounds were rare in the internal validation set. Specifically, nonconsensus occurred in 2 of 230 cases (0.87%) for GPT-5 with MDT and in-context learning, whereas no nonconsensus events were observed for other LLM-based prediction models (all 0/230). These findings indicate the high operational efficiency of the simulated MDT framework.</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Representative chat-based illustration of the multiagent multidisciplinary team consultation framework. Within the framework, a supervisor agent presents deidentified patient information and coordinates an iterative discussion among 3 role-specialized agents: a cardiovascular surgeon, an anesthesiologist, and a neurologist. Each specialist provides an estimated probability of PNC and a rationale, after which the supervisor summarizes the discussion and outputs the final prediction.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e91257_fig05.png"/></fig></sec><sec id="s3-4"><title>Word Frequency Analysis of Multiagent Rationales</title><p>Word frequency analysis of the summary rationales generated by the supervisor agent in the internal validation set was used as an exploratory analysis of the content emphasized in the model outputs. As shown in <xref ref-type="fig" rid="figure6">Figure 6</xref>, the incorporation of in-context learning appeared to shift the emphasis of the generated rationales. In the scenarios without in-context learning (<xref ref-type="fig" rid="figure6">Figures 6A and 6C</xref>), the rationales were distributed across a broader range of general patient characteristics. In contrast, when the LLMs were provided with population-level statistical summaries through in-context learning (<xref ref-type="fig" rid="figure6">Figures 6B and 6D</xref>), the generated rationales more frequently mentioned variables that were presented as statistically significant in the context, including intraoperative peak lactate level.</p><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Word frequency analysis of the supervisor agent&#x2019;s summary rationale. (A) GPT-5+no context+multidisciplinary team (MDT), (B) GPT-5+context+MDT, (C) DeepSeek-V3+no context+MDT, and (D) DeepSeek-V3+context+MDT. Variables are sorted by frequency in descending order.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e91257_fig06.png"/></fig></sec><sec id="s3-5"><title>Sensitivity Analysis Using the Selected Variables</title><p>To further assess the robustness of the primary findings, we performed a sensitivity analysis in the internal validation set using the same 15 predictors selected by LASSO regression for both ML- and LLM-based models. As summarized in Tables S4 and S5 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>, restricting the models to the LASSO-selected predictors affected the performance of ML and LLM models differently. Among the ML models, the random forest model showed improved discriminative ability compared with the primary analysis (AUC 0.8308, 95% CI 0.7489&#x2010;0.9128), suggesting that the LASSO method could reduce noise and avoid overfitting. In contrast, among the LLM configurations, the use of the MDT framework did not result in a statistically significant improvement in AUC over the corresponding non-MDT settings according to the DeLong test. Although GPT-5 with MDT and in-context learning remained the best-performing LLM configuration with the LASSO-selected variables (AUC 0.8117, 95% CI 0.7144&#x2010;0.9089), the incremental benefit of MDT discussion appeared attenuated after variable reduction. This finding suggests that when the patient information is greatly reduced, the additional value of multiagent discussion for LLM-based prediction may be weakened.</p></sec><sec id="s3-6"><title>Prospective Validation of LLM Prediction</title><p>The prospective validation demonstrated performance trends consistent with the internal validation. For LLMs in Table S6 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>, GPT-5 with MDT and in-context learning attained the highest AUC. However, for DeepSeek-V3, the MDT with in-context learning configuration yielded slightly lower predictive performance (AUC=0.8009) than the no MDT with in-context learning configuration (AUC=0.8147), suggesting that the multiagent enhancement framework may be model-dependent and influenced by architectural or tuning characteristics. Exploratory pairwise comparisons further showed that adding the MDT framework yielded numerically higher AUCs for GPT-5 both without in-context learning (0.7533 vs 0.7170; <italic>P</italic>=.55) and with in-context learning (0.8280 vs 0.7882; <italic>P</italic>=.21), whereas for DeepSeek-V3, the MDT framework increased AUC in the no-context setting (0.7366 vs 0.6615; <italic>P</italic>=.38) but not in the context setting (0.8009 vs 0.8147; <italic>P</italic>=.83). Nonconsensus events in the prospective cohort were likewise rare, occurring in 1 of 120 cases (0.83%) for DeepSeek-V3 with MDT and in-context learning, while no nonconsensus events were observed in the other MDT configurations (all 0/120). These findings underscore the need for tailored optimization when implementing collaborative LLM frameworks in clinical prediction tasks.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>In this study, we developed and validated an LLM-based multiagent framework inspired by multidisciplinary collaboration for predicting PNC in patients after ATAAD surgery. In this study setting, GPT-5 combined with both multiagent collaboration and in-context learning achieved the highest discriminative performance, suggesting that the integration of role specialization with population-level statistical context may improve the model&#x2019;s ability to capture complex risk factors. Notably, we further restricted model inputs to the 15 variables selected by LASSO. The names of these variables are listed in Table S7 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. The random forest model numerically outperformed the best-performing GPT-5 configuration (AUC of 0.8308 vs 0.8117), suggesting that traditional ML may derive greater benefit from noise reduction and feature selection when the prediction task is constrained to a smaller set of structured variables. In contrast, among the LLM configurations, the incremental benefit of MDT relative to the corresponding non-MDT setting was attenuated and no longer statistically significant. Although GPT-5 with both multiagent collaboration and in-context learning remained the best-performing LLM configuration in this setting, its advantage over the non-MDT mode was smaller than in the primary analysis. These findings suggest that the added value of multiagent collaboration may depend not only on the model itself but also on the richness of the available input information. When patient information is compressed into a smaller set of key variables, the amount of incremental information available for integration and discussion across roles is correspondingly reduced, thereby weakening the potential benefit of multiround collaboration. At the same time, the relatively low <italic>F</italic><sub>1</sub>-scores across models should be interpreted in the context of the low incidence of PNC in our cohort, which makes positive predictions intrinsically more difficult and implies that the models may be more suitable for risk stratification and early warning than for establishing a definitive positive diagnosis in isolation. Importantly, this benefit was not consistent across all backbone models. In the prospective validation cohort, GPT-5 maintained the best performance under the multirole setting, whereas the AUC of DeepSeek-V3 decreased from 0.8147 to 0.8009 after the addition of MDT, suggesting that the effect of multiagent collaboration is at least partly model-dependent. Another possible explanation is that the outcome-stratified in-context summaries were derived from the retrospective training cohort and may therefore have been less transportable to the shifted prospective population. This prior population mismatch may have contributed to the performance degradation of DeepSeek-V3 after adding MDT discussion in the in-context learning setting, particularly if the model overweighted retrospective cohort patterns that were less representative of the prospective cohort.</p></sec><sec id="s4-2"><title>Comparison With Prior Work</title><p>In recent years, LLMs have achieved near-human or even above-human performance on several national medical licensing examinations, suggesting that they encode a substantial amount of biomedical knowledge [<xref ref-type="bibr" rid="ref38">38</xref>]. However, such benchmarks are typically based on well-structured questions with a single best answer and therefore do not adequately capture the ambiguity, incomplete information, and longitudinal reasoning commonly encountered in real-world clinical practice. Prior studies have shown that when confronted with complex diagnostic cases or real-world prediction tasks, LLMs may still miss key differential diagnoses, and their calibration and awareness of uncertainty remain limited in patient-level clinical decision support [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref39">39</xref>].</p><p>To address these limitations, recent work has increasingly explored the use of LLMs within multiagent collaborative frameworks. For example, MedAgents improved zero-shot performance on medical question-answering tasks by organizing specialty-informed agents to debate and aggregate opinions [<xref ref-type="bibr" rid="ref40">40</xref>], while the clinically inspired multiagent conversational framework proposed by Chen et al [<xref ref-type="bibr" rid="ref29">29</xref>] also outperformed a single-agent baseline in rare disease diagnosis. At the same time, emerging evidence suggests that the advantage of multiagent collaboration is not stable across all tasks and models. Although multiagent frameworks may achieve better performance in complex, highly structured tasks, many problems can still be handled adequately by simpler single-agent strategies, often at substantially lower computational cost. Recent systematic evaluations of medical agent systems have similarly shown that their overall performance gains over baseline LLMs are modest and not consistently significant across scenarios, indicating that the practical value of multirole collaboration depends on the specific task, the underlying model, and the application context rather than representing a universally superior paradigm [<xref ref-type="bibr" rid="ref41">41</xref>]. In this study, the relatively small sample size and low event rate led to wide CIs, and most comparisons across LLM configurations did not reach statistical significance. Therefore, the proposed MDT-based framework should be interpreted primarily as a methodological proof-of-concept for structured multiagent risk assessment, and its clinical effectiveness requires further validation in larger prospective cohorts.</p><p>Against this background, our findings are broadly consistent with the existing literature. We constructed a multiagent framework composed of a cardiovascular surgeon, anesthesiologist, neurologist, and supervisor, and combined it with in-context learning based on population-level summary statistics. The results indicate that this framework can provide transparent, role-differentiated explanatory reasoning for PNC risk prediction and achieve the best discriminative performance with GPT-5, suggesting that structured collaborative reasoning may offer incremental benefit when paired with a sufficiently capable backbone model. However, the slight performance decline observed in DeepSeek-V3 after adding MDT in the prospective validation also indicates that this added value is model-dependent. One possible explanation is that the multirole framework in this study is built on a single shared backbone model, such that the different roles primarily represent distinct perspectives rather than truly independent information sources or heterogeneous knowledge systems. When the model is already able to make sufficient use of statistical context under the single-agent plus in-context learning setting, additional multiround collaboration may fail to generate meaningful new information and instead introduce coordination costs and informational redundancy, thereby diminishing the potential marginal gain. Taken together, these findings suggest that the primary value of multirole collaboration may lie in improving the structure and interpretability of the reasoning process, whereas gains in predictive performance remain contingent on the specific model and task setting [<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref42">42</xref>].</p><p>In addition, our findings indicate that priming LLMs with population-level summary statistics can effectively ground their reasoning in domain-specific evidence and thereby improve PNC prediction. For resource-constrained clinical settings, this in-context learning strategy offers an accessible route for adapting LLMs to specialized tasks. Unlike fine-tuning, it does not require additional model training or substantial computational resources [<xref ref-type="bibr" rid="ref43">43</xref>]. Unlike few-shot prompting, it allows more systematic cohort-level information to be introduced within a limited context window [<xref ref-type="bibr" rid="ref44">44</xref>]. Compared with retrieval-augmented generation, which depends heavily on the quality of the external knowledge base and retrieval performance, this approach is more direct and easier to implement [<xref ref-type="bibr" rid="ref45">45</xref>]. Thus, in-context learning based on population-level evidence may provide a practical pathway for enhancing the clinical utility of LLMs. However, we must acknowledge that several baseline and operative characteristics differed substantially between the retrospective and prospective cohorts, including arch replacement strategy and pericardial effusion grading. Although these variables were not retained among the 15 predictors selected via LASSO regression, their imbalance may still indicate broader distributional differences between cohorts. These shifts may reflect temporal changes in patient selection, perioperative evaluation, and surgical strategy, but systematic differences between retrospective electronic health record abstraction and prospective departmental data collection cannot be excluded. It should also be emphasized that, although the present multiagent framework draws on the collaborative concept of MDT, it remains, in essence, a role-based reasoning and result-integration process implemented within a single underlying model. Unlike a real clinical MDT, it does not involve physical examination, joint imaging review, longitudinal disease-course assessment, or interactive clinical judgment based on accumulated experience. Accordingly, it is more appropriately regarded as a structured approximation of multidisciplinary decision logic rather than a direct analog of actual MDT consultation. This distinction more accurately reflects the technical nature of the framework and helps clarify both its potential value and its practical boundaries in clinical application.</p></sec><sec id="s4-3"><title>Limitations</title><p>First, the composite outcome is heterogeneous, including stroke, cerebral hemorrhage, paraplegia, and coma. As these neurological complications may arise from different pathophysiological mechanisms, especially paraplegia versus cerebral events, this outcome definition may have introduced confounding into model development and interpretation. Second, although prospectively validated, the relatively small prospective cohort and limited number of PNC events may have reduced the statistical power of the prospective validation and constrained the stability of performance estimates. In addition, the prospective validation cohort included only patients with complete prespecified predictor variables, which may have introduced selection bias. External validation across different regions, patient populations, and health care systems is therefore needed to further establish its broader applicability. Third, the in-context learning strategy incorporated summary statistics derived from the training cohort, an approach that is not equivalent to standard ML training on individual-level labeled records. The word frequency analysis of the generated rationales was only an exploratory proxy for the content emphasized in the model outputs and should not be interpreted as a direct representation of the models&#x2019; internal reasoning or causal decision-making process. Moreover, because the in-context learning prompts explicitly included population-level summary statistics and intergroup <italic>P</italic> values, the increased frequency of certain terms may partly reflect the models&#x2019; reiterating variables emphasized in the provided context, rather than independent feature attribution by the LLMs. Fourth, our in-context learning prompt provided population-level summary statistics, including intergroup <italic>P</italic> values, but asked the model to express its final rationale in clinical terms without mentioning statistical significance. Although this design improved the readability and clinician-like style of the outputs, it may have reduced the direct traceability of how statistical information influenced the generated rationale; therefore, these explanations should be interpreted as clinically oriented summaries rather than fully transparent statistical reasoning. Fifth, the consensus criterion in the multiagent discussion was prespecified as a maximum probability range of 0.100 across specialist agents, meaning that the difference between the highest and lowest predicted probabilities could not exceed 10 percentage points. This threshold was chosen as a pragmatic operational criterion to balance near-agreement against unnecessarily prolonged discussion. However, no widely accepted standard currently exists for calibrating consensus thresholds in simulated LLM-MDT deliberation. Therefore, further efforts are needed to explore alternative probability-range cutoffs. Sixth, although MICE was used to handle missing data, the imputed values were averaged to generate a single complete dataset, rather than being analyzed separately with performance estimates pooled via the Rubin rules. This approach may have underestimated the uncertainty associated with missing-data imputation. In addition, imputed values were not explicitly labeled in the patient-specific prompts. Consequently, the LLMs treated them with the same confidence as observed values, which may have affected the models&#x2019; reasoning. Seventh, the operating thresholds used to calculate sensitivity, specificity, PPV, NPV, accuracy, and <italic>F</italic><sub>1</sub>-score were selected by maximizing Youden index within the same internal or prospective validation cohort in which these metrics were evaluated, rather than being prespecified and locked. Consequently, these threshold-dependent point estimates may be optimistically biased and should be interpreted as exploratory, cohort-optimized performance estimates rather than performance at a prespecified clinical operating threshold. This limitation does not affect the threshold-independent AUC comparisons or Brier scores.</p></sec><sec id="s4-4"><title>Conclusions</title><p>This study developed and validated an MDT-style LLM framework for predicting PNC after ATAAD surgery. The GPT-5 configuration combining multiagent collaboration with population-level in-context learning achieved the highest numerical discrimination among the evaluated LLM settings. However, adding the MDT framework did not produce a statistically significant improvement in AUC over the corresponding single-agent settings, and its incremental predictive value remains unconfirmed. The framework generated structured, role-specific rationales that were broadly consistent with clinically recognized risk factors, supporting its further investigation as a proof-of-concept approach for postoperative risk stratification and early warning. Larger multicenter external validation studies are required before routine clinical implementation.</p></sec></sec></body><back><ack><p>During the preparation of this manuscript, the authors used ChatGPT (OpenAI) for language polishing to enhance clarity and readability. The tool was used solely to aid in word choice and expression. All suggestions generated by AI were carefully reviewed and revised by the authors, who take full responsibility for the accuracy, integrity, and originality of the final work.</p></ack><notes><sec><title>Funding</title><p>This work was supported by grants from the National Natural Science Foundation of China (grant 82470386 to CQ).</p></sec><sec><title>Data Availability</title><p>The data used to support the findings of this study are available from the corresponding author upon request. No new software package or proprietary algorithm was developed in this study. The LLM-MDT (multidisciplinary team&#x2013;based large language model framework) workflow was implemented using custom Python scripts based on standard application programming interface (API) calls and the AutoGen framework, and the machine learning and statistical analyses were performed using standard Python and R packages. The complete prompt schemas and JSON contexts used for in-context learning are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. The custom scripts for prompt construction, multiagent orchestration, and model evaluation are available from the corresponding author upon reasonable request for academic and noncommercial use, subject to institutional data use and API access restrictions.</p></sec></notes><fn-group><fn fn-type="con"><p>JL, JZ, and CQ conceived the study. JL, JZ, XT, YC, SX, GQ, KD, and CQ performed the analysis, interpreted the results, and drafted the manuscript. All authors revised the manuscript and approved the final manuscript.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ATAAD</term><def><p>acute type A aortic dissection</p></def></def-item><def-item><term id="abb2">AUC</term><def><p>area under the receiver operating characteristic curve</p></def></def-item><def-item><term id="abb3">JSON</term><def><p>JavaScript Object Notation</p></def></def-item><def-item><term id="abb4">LASSO</term><def><p>least absolute shrinkage and selection operator</p></def></def-item><def-item><term id="abb5">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb6">MDT</term><def><p>multidisciplinary team</p></def></def-item><def-item><term id="abb7">MICE</term><def><p>multiple imputation by chained equations</p></def></def-item><def-item><term id="abb8">ML</term><def><p>machine learning</p></def></def-item><def-item><term id="abb9">NPV</term><def><p>negative predictive value</p></def></def-item><def-item><term id="abb10">PNC</term><def><p>postoperative neurological complication</p></def></def-item><def-item><term id="abb11">PPV</term><def><p>positive predictive value</p></def></def-item><def-item><term id="abb12">SHAP</term><def><p>Shapley Additive Explanations</p></def></def-item><def-item><term id="abb13">TRIPOD+AI</term><def><p>Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis + AI</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mikail</surname><given-names>N</given-names> </name><name name-style="western"><surname>Sablonier</surname><given-names>N</given-names> </name><name name-style="western"><surname>Gebert</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Age modulates the link between stress-related neural activity and mortality</article-title><source>Nat Commun</source><year>2025</year><month>11</month><day>7</day><volume>16</volume><issue>1</issue><fpage>9835</fpage><pub-id pub-id-type="doi">10.1038/s41467-025-64802-3</pub-id><pub-id pub-id-type="medline">41203612</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pena-Gralle</surname><given-names>APB</given-names> </name><name name-style="western"><surname>Forget</surname><given-names>A</given-names> </name><name name-style="western"><surname>Chiu</surname><given-names>YM</given-names> </name><name name-style="western"><surname>Legault</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Beauchesne</surname><given-names>MF</given-names> </name><name name-style="western"><surname>Blais</surname><given-names>L</given-names> </name></person-group><article-title>Medication-based mortality prediction in COPD using machine learning and conventional statistical methods</article-title><source>Int J Med Inform</source><year>2026</year><month>02</month><volume>206</volume><fpage>106177</fpage><pub-id pub-id-type="doi">10.1016/j.ijmedinf.2025.106177</pub-id><pub-id pub-id-type="medline">41202398</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Mao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name></person-group><article-title>Predicting mortality in intensive care unit patients with heart failure using an interpretable machine learning model: retrospective cohort study</article-title><source>J Med Internet Res</source><year>2022</year><month>08</month><day>9</day><volume>24</volume><issue>8</issue><fpage>e38082</fpage><pub-id pub-id-type="doi">10.2196/38082</pub-id><pub-id pub-id-type="medline">35943767</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yoo</surname><given-names>SK</given-names> </name><name name-style="western"><surname>Fitzgerald</surname><given-names>CW</given-names> </name><name name-style="western"><surname>Cho</surname><given-names>BA</given-names> </name><etal/></person-group><article-title>Prediction of checkpoint inhibitor immunotherapy efficacy for cancer using routine blood tests and clinical data</article-title><source>Nat Med</source><year>2025</year><month>03</month><volume>31</volume><issue>3</issue><fpage>869</fpage><lpage>880</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03398-5</pub-id><pub-id pub-id-type="medline">39762425</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lewin-Epstein</surname><given-names>O</given-names> </name><name name-style="western"><surname>Baruch</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hadany</surname><given-names>L</given-names> </name><name name-style="western"><surname>Stein</surname><given-names>GY</given-names> </name><name name-style="western"><surname>Obolski</surname><given-names>U</given-names> </name></person-group><article-title>Predicting antibiotic resistance in hospitalized patients by applying machine learning to electronic medical records</article-title><source>Clin Infect Dis</source><year>2021</year><month>06</month><day>1</day><volume>72</volume><issue>11</issue><fpage>e848</fpage><lpage>e855</lpage><pub-id pub-id-type="doi">10.1093/cid/ciaa1576</pub-id><pub-id pub-id-type="medline">33070171</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mintz</surname><given-names>I</given-names> </name><name name-style="western"><surname>Chowers</surname><given-names>M</given-names> </name><name name-style="western"><surname>Obolski</surname><given-names>U</given-names> </name></person-group><article-title>Prediction of ciprofloxacin resistance in hospitalized patients using machine learning</article-title><source>Commun Med (Lond)</source><year>2023</year><month>03</month><day>28</day><volume>3</volume><issue>1</issue><fpage>43</fpage><pub-id pub-id-type="doi">10.1038/s43856-023-00275-z</pub-id><pub-id pub-id-type="medline">36977789</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>C</given-names> </name><name name-style="western"><surname>Malin</surname><given-names>BA</given-names> </name><name name-style="western"><surname>Patel</surname><given-names>MB</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Y</given-names> </name></person-group><article-title>Predicting next-day discharge via electronic health record access logs</article-title><source>J Am Med Inform Assoc</source><year>2021</year><month>11</month><day>25</day><volume>28</volume><issue>12</issue><fpage>2670</fpage><lpage>2680</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocab211</pub-id><pub-id pub-id-type="medline">34592753</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>C</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Optimizing large language models for discharge prediction: best practices in leveraging electronic health record audit logs</article-title><source>AMIA Annu Symp Proc</source><year>2025</year><volume>2024</volume><fpage>1323</fpage><lpage>1331</lpage><pub-id pub-id-type="medline">40417553</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>AH</given-names> </name><name name-style="western"><surname>Chan</surname><given-names>MK</given-names> </name><name name-style="western"><surname>Narayanan</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Augmenting decision making in acute care surgery: a systematic review of machine learning-driven risk prediction models</article-title><source>J Trauma Acute Care Surg</source><year>2026</year><month>02</month><day>1</day><volume>100</volume><issue>2</issue><fpage>332</fpage><lpage>338</lpage><pub-id pub-id-type="doi">10.1097/TA.0000000000004805</pub-id><pub-id pub-id-type="medline">41196221</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>M</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name></person-group><article-title>Artificial intelligence for the diagnosis and management of cancers: potentials and challenges</article-title><source>MedComm (2020)</source><year>2025</year><month>11</month><volume>6</volume><issue>11</issue><fpage>e70460</fpage><pub-id pub-id-type="doi">10.1002/mco2.70460</pub-id><pub-id pub-id-type="medline">41200279</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name></person-group><article-title>Large language model&#x2013;based analysis of statin therapy discussions and sentiment on social media: cross-sectional observational study</article-title><source>J Med Internet Res</source><year>2026</year><month>04</month><day>10</day><volume>28</volume><fpage>e85057</fpage><pub-id pub-id-type="doi">10.2196/85057</pub-id><pub-id pub-id-type="medline">41962123</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sahoo</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Plasek</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Large language models for biomedicine: foundations, opportunities, challenges, and best practices</article-title><source>J Am Med Inform Assoc</source><year>2024</year><month>09</month><day>1</day><volume>31</volume><issue>9</issue><fpage>2114</fpage><lpage>2124</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocae074</pub-id><pub-id pub-id-type="medline">38657567</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gu</surname><given-names>B</given-names> </name><name name-style="western"><surname>Desai</surname><given-names>RJ</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>KJ</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>J</given-names> </name></person-group><article-title>Probabilistic medical predictions of large language models</article-title><source>NPJ Digit Med</source><year>2024</year><month>12</month><day>19</day><volume>7</volume><issue>1</issue><fpage>367</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01366-4</pub-id><pub-id pub-id-type="medline">39702641</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Large language model-based biological age prediction in large-scale populations</article-title><source>Nat Med</source><year>2025</year><month>09</month><volume>31</volume><issue>9</issue><fpage>2977</fpage><lpage>2990</lpage><pub-id pub-id-type="doi">10.1038/s41591-025-03856-8</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rao</surname><given-names>A</given-names> </name><name name-style="western"><surname>Pang</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Assessing the utility of ChatGPT throughout the entire clinical workflow: development and usability study</article-title><source>J Med Internet Res</source><year>2023</year><month>08</month><day>22</day><volume>25</volume><fpage>e48659</fpage><pub-id pub-id-type="doi">10.2196/48659</pub-id><pub-id pub-id-type="medline">37606976</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hirosawa</surname><given-names>T</given-names> </name><name name-style="western"><surname>Kawamura</surname><given-names>R</given-names> </name><name name-style="western"><surname>Harada</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>ChatGPT-generated differential diagnosis lists for complex case-derived clinical vignettes: diagnostic accuracy evaluation</article-title><source>JMIR Med Inform</source><year>2023</year><month>10</month><day>9</day><volume>11</volume><fpage>e48808</fpage><pub-id pub-id-type="doi">10.2196/48808</pub-id><pub-id pub-id-type="medline">37812468</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ben Shoham</surname><given-names>O</given-names> </name><name name-style="western"><surname>Rappoport</surname><given-names>N</given-names> </name></person-group><article-title>CPLLM: clinical prediction with large language models</article-title><source>PLOS Digit Health</source><year>2024</year><month>12</month><volume>3</volume><issue>12</issue><fpage>e0000680</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000680</pub-id><pub-id pub-id-type="medline">39642102</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gaber</surname><given-names>F</given-names> </name><name name-style="western"><surname>Shaik</surname><given-names>M</given-names> </name><name name-style="western"><surname>Allega</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Evaluating large language model workflows in clinical decision support for triage and referral and diagnosis</article-title><source>NPJ Digit Med</source><year>2025</year><month>05</month><day>9</day><volume>8</volume><issue>1</issue><fpage>263</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01684-1</pub-id><pub-id pub-id-type="medline">40346344</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Brown</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>C</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Large language models are less effective at clinical prediction tasks than locally trained machine learning models</article-title><source>J Am Med Inform Assoc</source><year>2025</year><month>05</month><day>1</day><volume>32</volume><issue>5</issue><fpage>811</fpage><lpage>822</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocaf038</pub-id><pub-id pub-id-type="medline">40056436</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kanjee</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Crowe</surname><given-names>B</given-names> </name><name name-style="western"><surname>Rodman</surname><given-names>A</given-names> </name></person-group><article-title>Accuracy of a generative artificial intelligence model in a complex diagnostic challenge</article-title><source>JAMA</source><year>2023</year><month>07</month><day>3</day><volume>330</volume><issue>1</issue><fpage>78</fpage><lpage>80</lpage><pub-id pub-id-type="doi">10.1001/jama.2023.8288</pub-id><pub-id pub-id-type="medline">37318797</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rylski</surname><given-names>B</given-names> </name><name name-style="western"><surname>Schilling</surname><given-names>O</given-names> </name><name name-style="western"><surname>Czerny</surname><given-names>M</given-names> </name></person-group><article-title>Acute aortic dissection: evidence, uncertainties, and future therapies</article-title><source>Eur Heart J</source><year>2023</year><month>03</month><day>7</day><volume>44</volume><issue>10</issue><fpage>813</fpage><lpage>821</lpage><pub-id pub-id-type="doi">10.1093/eurheartj/ehac757</pub-id><pub-id pub-id-type="medline">36540036</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Hemiarch repair versus total-arch repair with frozen elephant trunk for acute type A aortic dissection: A propensity score&#x2013;matched study</article-title><source>JTCVS Open</source><year>2026</year><month>06</month><fpage>101945</fpage><pub-id pub-id-type="doi">10.1016/j.xjon.2026.101945</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Evangelista</surname><given-names>A</given-names> </name><name name-style="western"><surname>Isselbacher</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Bossone</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Insights from the International Registry of Acute Aortic Dissection: a 20-year experience of collaborative clinical research</article-title><source>Circulation</source><year>2018</year><month>04</month><day>24</day><volume>137</volume><issue>17</issue><fpage>1846</fpage><lpage>1860</lpage><pub-id pub-id-type="doi">10.1161/CIRCULATIONAHA.117.031264</pub-id><pub-id pub-id-type="medline">29685932</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dumfarth</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kofler</surname><given-names>M</given-names> </name><name name-style="western"><surname>Stastny</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Immediate surgery in acute type A dissection and neurologic dysfunction: fighting the inevitable?</article-title><source>Ann Thorac Surg</source><year>2020</year><month>07</month><volume>110</volume><issue>1</issue><fpage>5</fpage><lpage>12</lpage><pub-id pub-id-type="doi">10.1016/j.athoracsur.2020.01.026</pub-id><pub-id pub-id-type="medline">32114042</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Demers</surname><given-names>P</given-names> </name><name name-style="western"><surname>Elkouri</surname><given-names>S</given-names> </name><name name-style="western"><surname>Martineau</surname><given-names>R</given-names> </name><name name-style="western"><surname>Couturier</surname><given-names>A</given-names> </name><name name-style="western"><surname>Cartier</surname><given-names>R</given-names> </name></person-group><article-title>Outcome with high blood lactate levels during cardiopulmonary bypass in adult cardiac operation</article-title><source>Ann Thorac Surg</source><year>2000</year><month>12</month><volume>70</volume><issue>6</issue><fpage>2082</fpage><lpage>2086</lpage><pub-id pub-id-type="doi">10.1016/s0003-4975(00)02160-3</pub-id><pub-id pub-id-type="medline">11156124</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Elbatarny</surname><given-names>M</given-names> </name><name name-style="western"><surname>Trimarchi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Korach</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Axillary vs femoral arterial cannulation in acute type A dissection: international multicenter data</article-title><source>Ann Thorac Surg</source><year>2024</year><month>06</month><volume>117</volume><issue>6</issue><fpage>1128</fpage><lpage>1134</lpage><pub-id pub-id-type="doi">10.1016/j.athoracsur.2024.02.026</pub-id><pub-id pub-id-type="medline">38458510</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Baudo</surname><given-names>M</given-names> </name><name name-style="western"><surname>D&#x2019;Alonzo</surname><given-names>M</given-names> </name><name name-style="western"><surname>Muneretto</surname><given-names>C</given-names> </name><name name-style="western"><surname>Benussi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Di Bacco</surname><given-names>L</given-names> </name><name name-style="western"><surname>Rosati</surname><given-names>F</given-names> </name></person-group><article-title>Unilateral vs. bilateral selective cerebral perfusion for acute type A aortic dissection with frozen elephant trunk: systematic review and meta-analysis</article-title><source>J Clin Med</source><year>2025</year><month>09</month><day>10</day><volume>14</volume><issue>18</issue><fpage>6392</fpage><pub-id pub-id-type="doi">10.3390/jcm14186392</pub-id><pub-id pub-id-type="medline">41010595</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hoefsmit</surname><given-names>PC</given-names> </name><name name-style="western"><surname>Schretlen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Does</surname><given-names>RJMM</given-names> </name><name name-style="western"><surname>Verouden</surname><given-names>NJ</given-names> </name><name name-style="western"><surname>Zandbergen</surname><given-names>HR</given-names> </name></person-group><article-title>Quality and process improvement of the multidisciplinary Heart Team meeting using Lean Six Sigma</article-title><source>BMJ Open Qual</source><year>2023</year><month>01</month><volume>12</volume><issue>1</issue><fpage>e002050</fpage><pub-id pub-id-type="doi">10.1136/bmjoq-2022-002050</pub-id><pub-id pub-id-type="medline">36707122</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>X</given-names> </name><name name-style="western"><surname>Yi</surname><given-names>H</given-names> </name><name name-style="western"><surname>You</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Enhancing diagnostic capability with multi-agents conversational large language models</article-title><source>NPJ Digit Med</source><year>2025</year><month>03</month><day>13</day><volume>8</volume><issue>1</issue><fpage>159</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01550-0</pub-id><pub-id pub-id-type="medline">40082662</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yin</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>P</given-names> </name></person-group><article-title>A proactive agent collaborative framework for zero-shot multimodal medical reasoning</article-title><source>Adv Intell Syst</source><year>2025</year><month>08</month><volume>7</volume><issue>8</issue><fpage>2400840</fpage><pub-id pub-id-type="doi">10.1002/aisy.202400840</pub-id><pub-id pub-id-type="medline">40852092</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Taberna</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gil Moncayo</surname><given-names>F</given-names> </name><name name-style="western"><surname>Jan&#x00E9;-Salas</surname><given-names>E</given-names> </name><etal/></person-group><article-title>The multidisciplinary team (MDT) approach and quality of care</article-title><source>Front Oncol</source><year>2020</year><volume>10</volume><fpage>85</fpage><pub-id pub-id-type="doi">10.3389/fonc.2020.00085</pub-id><pub-id pub-id-type="medline">32266126</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Miao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Luo</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name></person-group><article-title>MedARC: adaptive multi-agent refinement and collaboration for enhanced medical reasoning in large language models</article-title><source>Int J Med Inform</source><year>2026</year><month>02</month><volume>206</volume><fpage>106136</fpage><pub-id pub-id-type="doi">10.1016/j.ijmedinf.2025.106136</pub-id><pub-id pub-id-type="medline">41109093</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shaikh</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Jeelani-Shaikh</surname><given-names>ZA</given-names> </name><name name-style="western"><surname>Jeelani</surname><given-names>MM</given-names> </name><etal/></person-group><article-title>Collaborative intelligence in AI: evaluating the performance of a council of AIs on the USMLE</article-title><source>PLOS Digit Health</source><year>2025</year><month>10</month><volume>4</volume><issue>10</issue><fpage>e0000787</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000787</pub-id><pub-id pub-id-type="medline">41066364</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Collins</surname><given-names>GS</given-names> </name><name name-style="western"><surname>Moons</surname><given-names>KGM</given-names> </name><name name-style="western"><surname>Dhiman</surname><given-names>P</given-names> </name><etal/></person-group><article-title>TRIPOD+AI statement: updated guidance for reporting clinical prediction models that use regression or machine learning methods</article-title><source>BMJ</source><year>2024</year><month>04</month><day>16</day><volume>385</volume><fpage>e078378</fpage><pub-id pub-id-type="doi">10.1136/bmj-2023-078378</pub-id><pub-id pub-id-type="medline">38626948</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Lundberg</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>SI</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Guyon</surname><given-names>I</given-names> </name><name name-style="western"><surname>Von Luxburg</surname><given-names>U</given-names> </name><name name-style="western"><surname>Bengio</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wallach</surname><given-names>H</given-names> </name><name name-style="western"><surname>Fergus</surname><given-names>R</given-names> </name><name name-style="western"><surname>Vishwanathan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Garnett</surname><given-names>R</given-names> </name></person-group><article-title>A unified approach to interpreting model predictions</article-title><source>Advances in Neural Information Processing System</source><year>2017</year><access-date>2026-07-31</access-date><publisher-name>Neural Information Processing Systems Foundation</publisher-name><fpage>4765</fpage><lpage>4774</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2017/file/8a20a8621978632d76c43dfd28b67767-Paper.pdf">https://proceedings.neurips.cc/paper_files/paper/2017/file/8a20a8621978632d76c43dfd28b67767-Paper.pdf</ext-link></comment></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Bansal</surname><given-names>G</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Li</surname><given-names>B</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>E</given-names> </name><etal/></person-group><article-title>AutoGen: enabling next-gen LLM applications via multi-agent conversations</article-title><access-date>2026-07-19</access-date><conf-name>Presented at: First Conference on Language Modeling (COLM 2024)</conf-name><conf-date>Oct 7-9, 2024</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://openreview.net/pdf?id=BAakY1hNKS">https://openreview.net/pdf?id=BAakY1hNKS</ext-link></comment></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Van Buuren</surname><given-names>S</given-names> </name><name name-style="western"><surname>Groothuis-Oudshoorn</surname><given-names>K</given-names> </name></person-group><article-title>MICE: multivariate imputation by chained equations in R</article-title><source>J Stat Softw</source><year>2011</year><volume>45</volume><issue>3</issue><fpage>1</fpage><lpage>67</lpage><pub-id pub-id-type="doi">10.18637/jss.v045.i03</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kasagga</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sapkota</surname><given-names>A</given-names> </name><name name-style="western"><surname>Changaramkumarath</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Performance of ChatGPT and large language models on medical licensing exams worldwide: a systematic review and network meta-analysis with meta-regression</article-title><source>Cureus</source><year>2025</year><month>10</month><volume>17</volume><issue>10</issue><fpage>e94300</fpage><pub-id pub-id-type="doi">10.7759/cureus.94300</pub-id><pub-id pub-id-type="medline">41230320</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hager</surname><given-names>P</given-names> </name><name name-style="western"><surname>Jungmann</surname><given-names>F</given-names> </name><name name-style="western"><surname>Holland</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Evaluation and mitigation of the limitations of large language models in clinical decision-making</article-title><source>Nat Med</source><year>2024</year><month>09</month><volume>30</volume><issue>9</issue><fpage>2613</fpage><lpage>2622</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03097-1</pub-id><pub-id pub-id-type="medline">38965432</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Tang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zou</surname><given-names>A</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Z</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Ku</surname><given-names>LW</given-names> </name><name name-style="western"><surname>Martins</surname><given-names>A</given-names> </name><name name-style="western"><surname>Srikumar</surname><given-names>V</given-names> </name></person-group><article-title>MedAgents: large language models as collaborators for zero-shot medical reasoning</article-title><source>Findings of the Association for Computational Linguistics</source><year>2024</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>599</fpage><lpage>621</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.findings-acl.33</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Carrero</surname><given-names>ZI</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Benchmarking large language model-based agent systems for clinical decision tasks</article-title><source>NPJ Digit Med</source><year>2026</year><month>02</month><day>18</day><volume>9</volume><issue>1</issue><fpage>259</fpage><pub-id pub-id-type="doi">10.1038/s41746-026-02443-6</pub-id><pub-id pub-id-type="medline">41708802</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bousquet</surname><given-names>C</given-names> </name><name name-style="western"><surname>Beltramin</surname><given-names>D</given-names> </name></person-group><article-title>Advantages and inconveniences of a multi-agent large language model system to mitigate cognitive biases in diagnostic challenges</article-title><source>J Med Internet Res</source><year>2025</year><month>01</month><day>20</day><volume>27</volume><fpage>e69742</fpage><pub-id pub-id-type="doi">10.2196/69742</pub-id><pub-id pub-id-type="medline">39832364</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>K</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Gottweis</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Toward expert-level medical question answering with large language models</article-title><source>Nat Med</source><year>2025</year><month>03</month><volume>31</volume><issue>3</issue><fpage>943</fpage><lpage>950</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03423-7</pub-id><pub-id pub-id-type="medline">39779926</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cui</surname><given-names>H</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>J</given-names> </name><etal/></person-group><article-title>LLMs-based few-shot disease predictions using EHR: a novel approach combining predictive agent reasoning and critical agent instruction</article-title><source>AMIA Annu Symp Proc</source><year>2025</year><volume>2024</volume><fpage>319</fpage><lpage>328</lpage><pub-id pub-id-type="medline">40417470</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ke</surname><given-names>YH</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>L</given-names> </name><name name-style="western"><surname>Elangovan</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Retrieval augmented generation for 10 large language models and its generalizability in assessing medical fitness</article-title><source>NPJ Digit Med</source><year>2025</year><month>04</month><day>5</day><volume>8</volume><issue>1</issue><fpage>187</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01519-z</pub-id><pub-id pub-id-type="medline">40185842</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Supplementary tables, figures, examples, and data.</p><media xlink:href="jmir_v28i1e91257_app1.docx" xlink:title="DOCX File, 694 KB"/></supplementary-material><supplementary-material id="app2"><label>Checklist 1</label><p>TRIPOD+AI checklist.</p><media xlink:href="jmir_v28i1e91257_app2.pdf" xlink:title="PDF File, 773 KB"/></supplementary-material></app-group></back></article>