<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e101137</article-id><article-id pub-id-type="doi">10.2196/101137</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Large Language Models in Multidisciplinary Decision-Making for Hepatopancreatobiliary Oncology: Retrospective Comparative Feasibility Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Jo</surname><given-names>Sung Jun</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chong</surname><given-names>Eui Hyuk</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kang</surname><given-names>Incheon</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yang</surname><given-names>Seok Jeong</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kang</surname><given-names>Beodeul</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kim</surname><given-names>Jung Sun</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chon</surname><given-names>Hong Jae</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kwon</surname><given-names>Chang-il</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Sung</surname><given-names>Min Je</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Shin</surname><given-names>Suk Pyo</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>An</surname><given-names>Chansik</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Lim</surname><given-names>Ho Yeong</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yu</surname><given-names>Jeong-Sik</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Jang</surname><given-names>Sujin</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Im</surname><given-names>Jung Ho</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ko</surname><given-names>Kwang Hyun</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Lee</surname><given-names>Sung Hwan</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Surgery, CHA University Bundang Medical Center</institution><addr-line>59 Yatap-ro, Bundang-gu</addr-line><addr-line>Seongnam</addr-line><addr-line>Gyeonggi-do</addr-line><country>Republic of Korea</country></aff><aff id="aff2"><institution>Department of Medical Oncology, CHA University Bundang Medical Center</institution><addr-line>Seongnam</addr-line><addr-line>Gyeonggi-do</addr-line><country>Republic of Korea</country></aff><aff id="aff3"><institution>Department of Gastroenterology, CHA University Bundang Medical Center</institution><addr-line>Seongnam</addr-line><addr-line>Gyeonggi-do</addr-line><country>Republic of Korea</country></aff><aff id="aff4"><institution>Department of Radiology, CHA University Bundang Medical Center</institution><addr-line>Seongnam</addr-line><addr-line>Gyeonggi-do</addr-line><country>Republic of Korea</country></aff><aff id="aff5"><institution>Department of Nuclear Medicine, CHA University Bundang Medical Center</institution><addr-line>Seongnam</addr-line><addr-line>Gyeonggi-do</addr-line><country>Republic of Korea</country></aff><aff id="aff6"><institution>Department of Radiation Oncology, CHA University Bundang Medical Center</institution><addr-line>Seongnam</addr-line><addr-line>Gyeonggi-do</addr-line><country>Republic of Korea</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Stone</surname><given-names>Alicia</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Avula</surname><given-names>Ganesh Praneeth Roy</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Morris</surname><given-names>Kevin</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Luo</surname><given-names>Peng</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Sung Hwan Lee, MD, PhD, Department of Surgery, CHA University Bundang Medical Center, 59 Yatap-ro, Bundang-gu, Seongnam, Gyeonggi-do, Republic of Korea, 82 031-780-5000; <email>leeshmd77_ca@chamc.co.kr</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>17</day><month>9</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e101137</elocation-id><history><date date-type="received"><day>12</day><month>05</month><year>2026</year></date><date date-type="rev-recd"><day>25</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>26</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Sung Jun Jo, Eui Hyuk Chong, Incheon Kang, Seok Jeong Yang, Beodeul Kang, Jung Sun Kim, Hong Jae Chon, Chang-il Kwon, Min Je Sung, Suk Pyo Shin, Chansik An, Ho Yeong Lim, Jeong-Sik Yu, Sujin Jang, Jung Ho Im, Kwang Hyun Ko, Sung Hwan Lee. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 17.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e101137"/><abstract><sec><title>Background</title><p>Hepatopancreatobiliary (HPB) malignancies require complex treatment planning that often relies on multidisciplinary team (MDT) discussions. Large language models (LLMs) have recently been explored for clinical decision support, but their performance within real-world multidisciplinary decision environments remains unclear. In particular, the stability of LLM-generated recommendations&#x2014;that is, whether a model produces the same answer when given the same clinical input&#x2014;has rarely been examined.</p></sec><sec><title>Objective</title><p>This study aimed to evaluate the stability of treatment recommendations generated by contemporary LLMs when identical HPB cases are queried repeatedly, and their concordance with the treatment decisions reached at an institutional MDT conference.</p></sec><sec sec-type="methods"><title>Methods</title><p>This retrospective study included consecutive cases discussed at a single-center HPB MDT conference between September 1, 2024, and August 31, 2025. Standardized clinical case summaries derived from preconference documentation were provided to 4 LLMs (GPT-4o, GPT-5.2, Gemini 3 Pro, and Claude Sonnet 4.5) through their consumer web interfaces. Each model recommended a treatment among predefined MDT treatment options, and identical queries were repeated 4 times in separate sessions. Stability was quantified as the discordance rate relative to the initial response and, without privileging any single query, as the mean pairwise agreement and Fleiss &#x03BA; across the 4 iterations. Concordance with MDT decisions was assessed using both the initial and modal responses, together with Cohen &#x03BA; and class-wise <italic>F</italic><sub>1</sub>-scores.</p></sec><sec sec-type="results"><title>Results</title><p>A total of 107 MDT cases were analyzed. Stability differed significantly across models (<italic>P</italic>=.01). Gemini 3 Pro showed the lowest discordance rate (mean 12.8%, SD 2.3%) and the highest reference-free agreement (Fleiss &#x03BA;=0.737), whereas GPT-4o showed the highest discordance rate (mean 30.2%, SD 6.5%) and the lowest agreement (Fleiss &#x03BA;=0.430). Concordance with MDT decisions ranged from 48.6% to 72.9% using the initial response and from 66.3% to 74.5% using the modal response, and the highest-performing model differed between the 2 definitions. Class-wise <italic>F</italic><sub>1</sub>-score was consistently lower for surgery (0.400&#x2010;0.520) than for chemotherapy (0.621&#x2010;0.836). Complete discordance occurred in 17 of 107 (15.9%) cases and in none of the 31 anatomically unresectable cases (Fisher exact test, <italic>P</italic>=.003). Recurrent or on-treatment disease (adjusted odds ratio [OR] 5.40, 95% CI 1.65-17.68; <italic>P</italic>=.005), pancreatic tumor location (adjusted OR 7.37, 95% CI 2.03-26.78; <italic>P</italic>=.002), and low MDT agreement level (adjusted OR 10.33, 95% CI 1.54-69.38; <italic>P</italic>=.016) were independently associated with complete discordance.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>LLM-generated treatment recommendations demonstrated moderate alignment with MDT decisions in HPB oncology. Importantly, response stability varied substantially across models, indicating that concordance alone is insufficient for evaluating LLMs as clinical decision support tools. These findings suggest that LLMs may serve as a reasoning-support layer in MDT-like decision environments, but their response stability must be systematically characterized before clinical integration.</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>AI</kwd><kwd>natural language processing</kwd><kwd>clinical decision support</kwd><kwd>neoplasms</kwd><kwd>shared decision-making</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Hepatopancreatobiliary (HPB) malignancies represent some of the most complex diseases in oncology, often requiring careful integration of diagnostic imaging, pathology, surgical evaluation, and systemic therapy planning [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. Optimal management frequently depends on multidisciplinary decision-making involving hepatobiliary surgeons, gastroenterologists, medical oncologists, radiologists, and radiation oncologists [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref4">4</xref>]. Multidisciplinary team (MDT) discussions have, therefore, become an essential component of modern HPB cancer care [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref4">4</xref>-<xref ref-type="bibr" rid="ref6">6</xref>]. Previous studies have shown that MDT-based management can improve staging accuracy, optimize treatment selection, and enhance adherence to evidence-based guidelines [<xref ref-type="bibr" rid="ref5">5</xref>-<xref ref-type="bibr" rid="ref7">7</xref>]. However, MDT discussions are inherently resource-intensive and require coordinated input from multiple specialists [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. As the complexity and number of oncologic cases increase, maintaining efficient and consistent MDT decision-making can be challenging, particularly in settings with limited expert availability [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref8">8</xref>].</p><p>Recent advances in AI, particularly large language models (LLMs), have introduced new possibilities for supporting complex clinical decision processes [<xref ref-type="bibr" rid="ref9">9</xref>-<xref ref-type="bibr" rid="ref11">11</xref>]. Unlike traditional rule-based decision support systems, LLMs are capable of integrating heterogeneous information and generating structured reasoning in natural language [<xref ref-type="bibr" rid="ref9">9</xref>-<xref ref-type="bibr" rid="ref11">11</xref>]. In emerging agentic AI frameworks, LLMs are increasingly considered a central reasoning layer capable of orchestrating multiple information sources and decision pathways [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref12">12</xref>]. In such systems, the role of LLMs is not merely to provide isolated recommendations but to coordinate complex reasoning processes and integrate diverse inputs&#x2014;functions that conceptually resemble the orchestration process occurring in multidisciplinary clinical discussions [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref12">12</xref>].</p><p>Despite the rapidly growing interest in applying LLMs to clinical decision support, most existing studies have primarily evaluated the accuracy or concordance of LLM-generated recommendations compared with physician decisions or clinical guidelines [<xref ref-type="bibr" rid="ref10">10</xref>-<xref ref-type="bibr" rid="ref13">13</xref>]. Although these studies provide important insights into the decision-making capabilities of LLMs, they do not fully address whether LLMs can function within complex clinical decision-making environments, such as MDT discussions [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref14">14</xref>]. In particular, little is known about the stability of LLM-generated recommendations when presented with identical clinical information, or about the degree to which such recommendations align with real-world MDT decisions [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref14">14</xref>].</p><p>Evidence from adjacent applications of LLMs to clinical text frames what can reasonably be expected in this setting. Validation studies of information extraction from oncology records report accuracy that is high but variable across document types and terminology [<xref ref-type="bibr" rid="ref15">15</xref>], establishing that these models can parse clinical narratives while leaving open whether the judgment built on that parsing is sound. The experience of deployed clinical prediction models is instructive in this regard: a widely implemented proprietary sepsis prediction model performed substantially worse on external validation at a single center than its development data had suggested [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>], indicating that performance must be established in the institution where a tool would be used rather than assumed from reported figures. Tumor board concordance studies in breast and primary liver cancer have begun this work [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>], but most rely on curated case profiles or a single query per case, so response stability and institution-level validity remain largely unexamined. Reporting standards for this class of study have only recently been formalized [<xref ref-type="bibr" rid="ref20">20</xref>]. Therefore, this study aimed to investigate the feasibility of LLMs as a potential orchestration component in multidisciplinary oncologic decision-making. Using real-world cases discussed in an HPB MDT conference, we conducted a comparative evaluation across 4 contemporary LLMs.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design and Population</title><p>This retrospective observational study evaluated the feasibility of LLMs as a potential orchestration component in multidisciplinary decision-making for HPB malignancies. All consecutive cases discussed at the HPB MDT conference at our institution between September 1, 2024, and August 31, 2025, were screened for eligibility. The MDT conference included specialists from hepatobiliary surgery, gastroenterology, medical oncology, radiology, and radiation oncology, and treatment decisions were determined through multidisciplinary discussion.</p><p>Cases were eligible for inclusion if they involved patients with HPB malignancies undergoing MDT discussion for treatment planning. Cases were excluded if they met any of the following criteria: (1) cases discussed solely for biliary drainage without a definitive oncologic diagnosis, (2) patients with synchronous or metachronous double primary malignancies, or (3) metastases to the HPB system from a non-HPB primary tumor. The final study cohort consisted of cases meeting the eligibility criteria during the study period.</p></sec><sec id="s2-2"><title>Clinical Variables and Definitions</title><p>Clinical variables used in the analysis were extracted from MDT records and electronic medical records. Age and sex were obtained from patient demographic data. Tumor markers, including carcinoembryonic antigen and carbohydrate antigen 19&#x2010;9, were recorded based on the laboratory results available at the time of the MDT discussion. Disease status was classified as primary disease, recurrent disease, or on-treatment disease. Tumor location was categorized as biliary tract, pancreas, or ampulla of Vater based on the primary tumor origin. Biliary tract cancers included intrahepatic cholangiocarcinoma, perihilar cholangiocarcinoma, distal common bile duct cancer, and gallbladder cancer. For pancreatic cancer cases, tumor location was further classified as head, neck, body, tail, or diffuse involvement according to radiologic reports. The main referring department was categorized as gastroenterology, medical oncology, or surgical oncology. The MDT agreement level was recorded based on the documented MDT conference conclusions and categorized as high confidence, moderate confidence, or low confidence, reflecting the degree of consensus among participating specialists. Caregiver support was determined based on the family members documented as accompanying the patient to the MDT conference, together with the patient&#x2019;s recorded residential address. Support was classified as present when a spouse attended. When the accompanying relative was an adult child or other family member, support was classified as present unless the patient resided in a rural area distant from that relative, in which case it was classified as absent. Cases with no accompanying family member were classified as absent.</p><p>Treatment recommendations were categorized as surgery, chemotherapy, radiation therapy, active surveillance, or further evaluation. Sequential and combined plans were assigned to the modality initiated first, so that a plan of neoadjuvant chemotherapy followed by reassessment for resection was categorized as chemotherapy. The same rules were applied to the documented MDT conclusion and to the model output. Full mapping rules and the institutional protocols in effect during the study period are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>Anatomic resectability was assessed retrospectively for all cases by 3 hepatobiliary and pancreatic surgeons reading together in a single consensus session. Assessors reviewed the cross-sectional imaging and classified each case as unresectable, borderline resectable, or resectable on the basis of the anatomic relationship between the tumor and the vasculature, with disagreements resolved by discussion until a single classification was agreed upon. Assessors were blinded to both the MDT decision and all model output. Because classification was reached by consensus rather than by independent reading, interrater agreement could not be calculated.</p></sec><sec id="s2-3"><title>Clinical Information Used as Model Input</title><p>Clinical information presented during the MDT conference was compiled and used as input for the LLMs. The case summaries were derived from the referral and preconference documentation prepared when the patient entered the MDT workflow, before the conference and before any treatment decision was made. For this study, the documentation was reformatted into a standardized template without alteration of clinical content. The documented MDT conclusion was not included in the model input, and only information available at the time of the conference was used. The input data included patient demographics (age and sex), radiologic findings from abdominal computed tomography or magnetic resonance imaging and chest computed tomography, histopathologic results from biopsy, when available, positron emission tomography findings, and laboratory data, including total bilirubin, aspartate aminotransferase, alanine aminotransferase, and tumor markers such as carbohydrate antigen 19&#x2010;9 and carcinoembryonic antigen. Radiologic and pathologic information was summarized in text format based on formal clinical reports rather than raw imaging or laboratory datasets. All clinical information was organized into a standardized textual case summary before being provided to the models.</p></sec><sec id="s2-4"><title>AI Models and Query Framework</title><p>Four contemporary LLMs were evaluated in this study: GPT-4o, GPT-5.2, Gemini 3 Pro, and Claude Sonnet 4.5. All models were accessed using their publicly available versions without additional fine-tuning. Each model received identical clinical case summaries and was asked to recommend the most appropriate treatment strategy among predefined options reflecting common MDT decisions: surgery, chemotherapy, radiation therapy, active surveillance, or further evaluation. The models were instructed to provide a clear recommendation with a brief explanation.</p><p>All queries were performed through the consumer web interfaces (ChatGPT, the Gemini app, and claude.ai) rather than provider APIs. Models were queried in separate blocks between November 2025 and May 2026: GPT-4o in November-December 2025, GPT-5.2 in December 2025-January 2026, Gemini 3 Pro in February 2026, and Claude Sonnet 4.5 over approximately 1 month thereafter. Query blocks were bounded by the periods during which each model was selectable in its consumer interface; exact query dates were not logged prospectively. The model was selected explicitly in the model picker for every query rather than relying on the application default.</p><p>GPT-5.2 was used in the default interface configuration in which the provider&#x2019;s router determines whether a given query is served by the Instant or the Thinking variant; this routing is not visible or user-configurable, and neither Thinking nor Pro was manually selected. More generally, these interfaces do not expose decoding parameters, so temperature, top-p, and system-level instructions remained at provider defaults. Browsing and retrieval were disabled, and account-level memory and personalization were turned off for all providers. Each of the 4 repeated queries for a given case was submitted in a new session, and each session was deleted after the response was recorded, so that no query and no case could enter the context of a subsequent one. Exact query dates were, therefore, not retained in the interface history and are reported at the level of the query block.</p><p>To evaluate response stability, identical queries were submitted to each model 4 times using the same clinical input and prompt. LLMs generate responses through probabilistic processes, which may produce variable outputs even when identical inputs are provided. Repeated queries were, therefore, used to assess the consistency of treatment recommendations under identical clinical conditions.</p></sec><sec id="s2-5"><title>Qualitative Review of Model Rationales</title><p>In addition to the categorical recommendation, the free-text rationale accompanying each response was retained. In a post hoc analysis, 4 cases were purposively selected for detailed review: 2 discordant cases representing opposite directions of error across the surgical decision boundary and 2 concordant cases as comparators. Selection was purposive rather than random, and no quantitative inference was drawn from these cases. Each rationale was reviewed against the corresponding MDT record by the study investigators. The annotated cases are presented in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></sec><sec id="s2-6"><title>Study Outcomes</title><p>The primary outcomes of this study were the stability of LLM-generated treatment recommendations and their clinical concordance with MDT decisions. Stability was defined as the consistency of treatment recommendations generated by each LLM when identical clinical information was provided repeatedly. To assess this, identical queries were submitted 4 times to each model using the same clinical case summary. Concordance was defined as agreement between the treatment recommendation generated by the LLM and the final treatment decision determined during the MDT conference. Concordance analysis was performed using both the first response and the modal response across the 4 repeated queries. The secondary outcome was to identify clinical factors associated with complete discordance, defined as cases in which all evaluated LLM models generated treatment recommendations that differed from the MDT decision.</p></sec><sec id="s2-7"><title>Statistical Analysis</title><p>Continuous variables are presented as mean (SD) or median (IQR), as appropriate, and categorical variables are presented as numbers and percentages. Comparisons between groups were performed using the 2-tailed Student <italic>t</italic> test or Mann-Whitney <italic>U</italic> test for continuous variables and the chi-square test or Fisher exact test for categorical variables, as appropriate. To evaluate the stability of LLM-generated treatment recommendations, discordance rates were calculated by comparing treatment categories from repeated queries (queries 2&#x2010;4) with the initial response (query 1) for each model. Because this definition designates the initial response as the reference, agreement across the 4 queries was additionally quantified without privileging any single query, using mean pairwise agreement across all 6 query pairs and Fleiss &#x03BA;, treating the 4 iterations as raters. Bootstrap 95% CIs for Fleiss &#x03BA; were obtained from 4000 resamples. Concordance with MDT decisions was initially assessed using the first response of each model. However, using a single response as the reference may misrepresent a model whose first output is atypical, since a recommendation that differs from the first response is counted as discordant, even when it is the recommendation the model produces most often. Concordance was, therefore, also assessed using the modal response, defined as the treatment category most frequently recommended across the 4 repeated queries; cases without a unique mode were excluded (n=6, 10, 5, and 9 for GPT-4o, GPT-5.2, Gemini 3 Pro, and Claude Sonnet 4.5, respectively). Under each definition, overall concordance, Cohen &#x03BA;, and class-wise <italic>F</italic><sub>1</sub>-scores were calculated against the MDT decision. Class-wise <italic>F</italic><sub>1</sub>-score was calculated only for treatment categories represented by at least 5 MDT decisions, since precision and recall are unstable below this frequency; radiation therapy (n=1), active surveillance (n=2), and further evaluation (n=2) fell below this threshold, so <italic>F</italic><sub>1</sub>-score is reported for surgery (n=25) and chemotherapy (n=77) only. Discordance rates among models were compared using the chi-square test. To identify clinical factors associated with complete discordance, univariate logistic regression analysis was performed. Variables for the multivariate model were prespecified on clinical grounds rather than selected by univariate significance and were limited to 3 given 17 events; Firth&#x2019;s penalized maximum likelihood was used because of the small number of events. Odds ratios (ORs) with 95% CIs were calculated. All statistical analyses were performed using R software (The R Core Team). <italic>P</italic>&#x003C;.05 (2-sided) was considered statistically significant.</p></sec><sec id="s2-8"><title>Ethical Considerations</title><p>The study protocol conformed to the ethical guidelines of the Declaration of Helsinki and was approved by the institutional review board of Bundang CHA Medical Center (IRB 2026-03-057). The requirement for informed consent was waived owing to the retrospective nature of the study. All patient data were deidentified before analysis.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Patient Characteristics</title><p>The flow diagram of patient inclusion is shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>. Between September 1, 2024, and August 31, 2025, a total of 113 cases were discussed at the HPB MDT conference and were screened for eligibility. After applying the predefined exclusion criteria&#x2014;including cases discussed solely for biliary drainage, patients with double primary malignancies, and metastases from non-HPB primary tumors&#x2014;107 cases were included in the final analysis.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Flow diagram of patient selection. HPB: hepatopancreatobiliary; LLM: large language model; MDT: multidisciplinary team.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e101137_fig01.png"/></fig><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Characteristics of patients.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Variables</td><td align="left" valign="bottom">Primary (n=77)</td><td align="left" valign="bottom">Recurrent or on-treatment (n=30)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Age (y), mean (SD)</td><td align="left" valign="top">66.4 (9.6)</td><td align="left" valign="top">65.0 (11.0)</td><td align="left" valign="top">.66</td></tr><tr><td align="left" valign="top" colspan="3">Sex, n (%)</td><td align="left" valign="top">.66</td></tr><tr><td align="left" valign="top">&#x2003;Male</td><td align="left" valign="top">46 (59.7)</td><td align="left" valign="top">20 (66.7)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">&#x2003;Female</td><td align="left" valign="top">31 (40.3)</td><td align="left" valign="top">10 (33.3)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="3">Main department, n (%)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">&#x2003;Gastroenterology</td><td align="left" valign="top">59 (76.6)</td><td align="left" valign="top">4 (13.3)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">&#x2003;Medical oncology</td><td align="left" valign="top">3 (3.9)</td><td align="left" valign="top">23 (76.7)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">&#x2003;Surgical oncology</td><td align="left" valign="top">15 (19.5)</td><td align="left" valign="top">3 (10.0)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="3">Tumor location (organ), n (%)</td><td align="left" valign="top">&#x003E;.99</td></tr><tr><td align="left" valign="top">&#x2003;Biliary tract</td><td align="left" valign="top">51 (66.2)</td><td align="left" valign="top">20 (66.7)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">&#x2003;Pancreas</td><td align="left" valign="top">24 (31.2)</td><td align="left" valign="top">9 (30.0)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">&#x2003;Ampulla of Vater</td><td align="left" valign="top">2 (2.6)</td><td align="left" valign="top">1 (3.3)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="3">Bile duct cancer type, n (%)<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="top">.07</td></tr><tr><td align="left" valign="top">&#x2003;Intrahepatic</td><td align="left" valign="top">12 (23.5)</td><td align="left" valign="top">5 (25)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">&#x2003;Hilar (perihilar)</td><td align="left" valign="top">30 (58.8)</td><td align="left" valign="top">6 (30)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">&#x2003;Distal common bile duct</td><td align="left" valign="top">5 (9.8)</td><td align="left" valign="top">5 (25)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">&#x2003;Gallbladder</td><td align="left" valign="top">4 (7.8)</td><td align="left" valign="top">4 (20)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="3">Pancreatic cancer site, n (%)<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="top">.52</td></tr><tr><td align="left" valign="top">&#x2003;Head</td><td align="left" valign="top">16 (66.7)</td><td align="left" valign="top">4 (44.4)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">&#x2003;Neck</td><td align="left" valign="top">1 (4.2)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">&#x2003;Body</td><td align="left" valign="top">3 (12.5)</td><td align="left" valign="top">3 (33.3)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">&#x2003;Tail</td><td align="left" valign="top">3 (12.5)</td><td align="left" valign="top">1 (11.1)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">&#x2003;Diffuse</td><td align="left" valign="top">1 (4.2)</td><td align="left" valign="top">1 (11.1)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="3">Anatomic resectability, n (%)</td><td align="left" valign="top">.28</td></tr><tr><td align="left" valign="top">&#x2003;Unresectable</td><td align="left" valign="top">20 (26.0)</td><td align="left" valign="top">11 (36.7)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">&#x2003;Borderline resectable</td><td align="left" valign="top">33 (42.9)</td><td align="left" valign="top">8 (26.7)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">&#x2003;Resectable</td><td align="left" valign="top">24 (31.2)</td><td align="left" valign="top">11 (36.7)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="3">MDT<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> decision, n (%)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">&#x2003;Surgery</td><td align="left" valign="top">12 (15.6)</td><td align="left" valign="top">13 (43.3)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">&#x2003;Chemotherapy</td><td align="left" valign="top">64 (83.1)</td><td align="left" valign="top">13 (43.3)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">&#x2003;Radiation therapy</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">1 (3.3)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">&#x2003;Active surveillance</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">2 (6.7)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">&#x2003;Further evaluation</td><td align="left" valign="top">1 (1.3)</td><td align="left" valign="top">1 (3.3)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="3">MDT agreement level, n (%)</td><td align="left" valign="top">.32</td></tr><tr><td align="left" valign="top">&#x2003;High confidence</td><td align="left" valign="top">66 (85.7)</td><td align="left" valign="top">22 (73.3)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">&#x2003;Moderate confidence</td><td align="left" valign="top">7 (9.1)</td><td align="left" valign="top">5 (16.7)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">&#x2003;Low confidence</td><td align="left" valign="top">4 (5.2)</td><td align="left" valign="top">3 (10.0)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top" colspan="4">Tumor marker</td></tr><tr><td align="left" valign="top">&#x2003;CEA<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup> (ng/mL), median (IQR)</td><td align="left" valign="top">2.81 (2.21&#x2010;4.33)</td><td align="left" valign="top">2.75 (1.96&#x2010;3.93)</td><td align="left" valign="top">.71</td></tr><tr><td align="left" valign="top">&#x2003;CA 19&#x2010;9<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup> (U/mL), median (IQR)</td><td align="left" valign="top">143.5 (45.0&#x2010;400.0)</td><td align="left" valign="top">37.8 (15.2&#x2010;267.8)</td><td align="left" valign="top">.03</td></tr><tr><td align="left" valign="top" colspan="3">Caregiver support, n (%)</td><td align="left" valign="top">&#x003E;.99</td></tr><tr><td align="left" valign="top">&#x2003;Present</td><td align="left" valign="top">72 (93.5)</td><td align="left" valign="top">28 (93.3)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top">&#x2003;Absent</td><td align="left" valign="top">5 (6.5)</td><td align="left" valign="top">2 (6.7)</td><td align="left" valign="top"/></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Percentages for bile duct cancer type are calculated among patients with biliary tract cancer (n=51 in the primary group and n=20 in the recurrent or on-treatment group), and percentages for pancreatic cancer site among patients with pancreatic cancer (n=24 and n=9, respectively). All other percentages are calculated within the full column.</p></fn><fn id="table1fn2"><p><sup>b</sup>MDT: multidisciplinary team.</p></fn><fn id="table1fn3"><p><sup>c</sup>CEA: carcinoembryonic antigen.</p></fn><fn id="table1fn4"><p><sup>d</sup>CA 19-9: carbohydrate antigen 19-9.</p></fn></table-wrap-foot></table-wrap><p>The baseline characteristics of the study population, stratified by disease status, are summarized in <xref ref-type="table" rid="table1">Table 1</xref>. Among the 107 included cases, 77 (72.0%) had primary disease, whereas 30 (28.0%) had recurrent or on-treatment disease.</p><p>The mean age was 66.4 (SD 9.6) years in the primary disease group and 65.0 (SD 11.0) years in the recurrent or on-treatment group (<italic>P</italic>=.66), and the proportion of male patients was similar between groups (46/77, 59.7% vs 20/30, 66.7%; <italic>P</italic>=.66). Regarding tumor location, the distribution of biliary tract, pancreas, and ampulla of Vater tumors was comparable between groups (<italic>P</italic>&#x003E;.99). However, the main referring department differed significantly, with gastroenterology accounting for the majority of primary disease cases (59/77, 76.6%), whereas medical oncology accounted for most recurrent or on-treatment cases (23/30, 76.7%; <italic>P</italic>&#x003C;.001). In terms of MDT treatment decisions, chemotherapy was the predominant recommendation in the primary disease group (64/77, 83.1%), whereas surgery and chemotherapy were equally recommended (13/30, 43.3% each) in the recurrent or on-treatment group (<italic>P</italic>&#x003C;.001). The distribution of MDT agreement levels and most tumor markers were comparable between groups, although CA 19&#x2010;9 levels were higher in the primary disease group (primary: median 143.5, IQR 45.0&#x2010;400.0 vs recurrent or on-treatment: median 37.8, IQR 15.2&#x2010;267.8; <italic>P</italic>=.03).</p></sec><sec id="s3-2"><title>Stability of LLM Responses</title><p>The stability of LLM-generated treatment recommendations was assessed by calculating discordance rates across repeated queries. For each model, treatment recommendations from repeated queries (queries 2&#x2010;4) were compared with the initial response (query 1). The mean discordance rates differed among the evaluated models. Gemini 3 Pro demonstrated the lowest discordance rate (mean 12.8%, SD 2.3%), followed by GPT-5.2 (mean 22.7%, SD 5.8%), Claude Sonnet 4.5 (mean 28.0%, SD 1.5%), and GPT-4o (mean 30.2%, SD 6.5%). The difference in discordance rates among models was statistically significant (<italic>P</italic>=.01). Detailed discordance rates for each repeated query are presented in <xref ref-type="table" rid="table2">Table 2</xref>.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Response stability and concordance of large language models (LLMs) with multidisciplinary team (MDT) decisions for hepatopancreatobiliary (HPB) treatment recommendations<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom" colspan="4">Discordance vs initial response (n=107)</td><td align="left" valign="bottom" colspan="2">Reference-free agreement</td><td align="left" valign="bottom" colspan="3">Concordance with MDT decision</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">#2, n (%)</td><td align="left" valign="bottom">#3, n (%)</td><td align="left" valign="bottom">#4, n (%)</td><td align="left" valign="bottom">Mean (SD)</td><td align="left" valign="bottom">Pairwise agreement (%)</td><td align="left" valign="bottom">Fleiss &#x03BA; (95% CI)</td><td align="left" valign="bottom">Concordance (n=107), n (%)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score for surgery</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score for chemotherapy</td></tr></thead><tbody><tr><td align="left" valign="top">GPT-4o</td><td align="left" valign="top">26 (24.3)</td><td align="left" valign="top">42 (39.3)</td><td align="left" valign="top">29 (27.1)</td><td align="left" valign="top">30.2 (6.5)</td><td align="left" valign="top">67.8</td><td align="left" valign="top">0.430 (0.33&#x2010;0.52)</td><td align="left" valign="top">78 (72.9)</td><td align="left" valign="top">0.520</td><td align="left" valign="top">0.836</td></tr><tr><td align="left" valign="top">GPT-5.2</td><td align="left" valign="top">21 (19.6)</td><td align="left" valign="top">33 (30.8)</td><td align="left" valign="top">19 (17.8)</td><td align="left" valign="top">22.7 (5.8)</td><td align="left" valign="top">74.5</td><td align="left" valign="top">0.535 (0.43&#x2010;0.63)</td><td align="left" valign="top">73 (68.2)</td><td align="left" valign="top">0.510</td><td align="left" valign="top">0.803</td></tr><tr><td align="left" valign="top">Gemini 3 Pro</td><td align="left" valign="top">11 (10.3)</td><td align="left" valign="top">13 (12.1)</td><td align="left" valign="top">17 (15.9)</td><td align="left" valign="top">12.8 (2.3)</td><td align="left" valign="top">87.1</td><td align="left" valign="top">0.737 (0.64&#x2010;0.82)</td><td align="left" valign="top">72 (67.3)</td><td align="left" valign="top">0.444</td><td align="left" valign="top">0.770</td></tr><tr><td align="left" valign="top">Claude Sonnet 4.5</td><td align="left" valign="top">30 (28.0)</td><td align="left" valign="top">28 (26.2)</td><td align="left" valign="top">32 (29.9)</td><td align="left" valign="top">28.0 (1.5)</td><td align="left" valign="top">74.0</td><td align="left" valign="top">0.572 (0.48&#x2010;0.66)</td><td align="left" valign="top">52 (48.6)</td><td align="left" valign="top">0.400</td><td align="left" valign="top">0.621</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Discordance rates were calculated by comparing the treatment category from each repeated query (#2 to #4) with the initial response (#1), and differed significantly across the 4 models (chi-square test, <italic>P</italic>=.01). Pairwise agreement is the mean agreement across all 6 query pairs, and Fleiss &#x03BA; treats the 4 iterations as raters. Concordance and <italic>F</italic><sub>1</sub>-scores are based on the initial response, with the MDT decision as the reference standard. <italic>F</italic><sub>1</sub>-score is reported for surgery and chemotherapy only.</p></fn></table-wrap-foot></table-wrap><p>Because these discordance rates used the initial response as the reference, agreement across the 4 queries was additionally quantified without designating any single query as the reference. Mean pairwise agreement across all 6 query pairs and Fleiss &#x03BA;, treating the 4 iterations as raters, were 67.8% and 0.430 for GPT-4o, 74.5% and 0.535 for GPT-5.2, 87.1% and 0.737 for Gemini 3 Pro, and 74.0% and 0.572 for Claude Sonnet 4.5. Gemini 3 Pro was the most stable model under every measure. The ordering of the remaining models was not preserved: Claude Sonnet 4.5 showed a higher discordance rate than GPT-5.2 (mean 28.0%, SD 1.5% vs mean 22.7%, SD 5.8%) but also higher reference-free agreement (Fleiss &#x03BA; 0.572 vs 0.535).</p></sec><sec id="s3-3"><title>Concordance With MDT Decisions</title><p>The concordance between LLM-generated treatment recommendations and MDT decisions across repeated queries is shown in <xref ref-type="fig" rid="figure2">Figure 2</xref>. In the first iteration (consisting of 107 cases), concordance rates were 72.9% (n=78) for GPT-4o, 68.2% (n=73) for GPT-5.2, 67.3% (n=72) for Gemini 3 Pro, and 48.6% (n=52) for Claude Sonnet 4.5. Across repeated queries, Gemini 3 Pro showed consistently high concordance rates (67.3%, 74.8%, 74.8%, and 72.9%), whereas GPT-4o, GPT-5.2, and Claude Sonnet 4.5 showed greater variability across iterations. When averaged across all 4 iterations, the mean concordance rates were 72.4% (SD 3.1%) for Gemini 3 Pro, 64.3% (SD 6.4%) for GPT-4o, 62.9% (SD 6.7%) for GPT-5.2, and 59.3% (SD 7.8%) for Claude Sonnet 4.5.</p><p>Class-wise performance based on the first response is shown in <xref ref-type="table" rid="table2">Table 2</xref>. Across all models, <italic>F</italic><sub>1</sub>-scores were consistently higher for chemotherapy (0.621&#x2010;0.836) than for surgery (0.400&#x2010;0.520), indicating that the models identified candidates for systemic therapy more reliably than candidates for surgery.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Concordance between large language model (LLM) recommendations and multidisciplinary team (MDT) decisions. (A) Concordance rates between LLM-generated treatment recommendations and MDT decisions across 4 repeated queries for each model. (B) Mean concordance rates averaged across the 4 iterations for each model. Error bars represent SD.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e101137_fig02.png"/></fig></sec><sec id="s3-4"><title>Concordance Using the First Response vs the Modal Response</title><p>Because concordance was assessed using the first response of each model, an additional analysis was performed using the modal response across the 4 repeated queries (<xref ref-type="table" rid="table3">Table 3</xref>). The ranking of models differed between the 2 reference responses. Using the first response, GPT-4o showed the highest concordance (78/107, 72.9%) and Claude Sonnet 4.5 the lowest (52/107, 48.6%), whereas, using the modal response, Gemini 3 Pro showed the highest concordance (76/102, 74.5%). The difference between the 2 reference responses was largest for Claude Sonnet 4.5, in which concordance increased from 52 of 107 (48.6%) to 65 of 98 (66.3%) and Cohen &#x03BA; from 0.115 to 0.356, and it was smallest for GPT-5.2 (73/107, 68.2% and 66/97, 68.0%).</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Concordance with multidisciplinary team (MDT) decisions using the first response vs the modal response.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom" colspan="2">Concordance</td><td align="left" valign="bottom" colspan="2">Cohen &#x03BA;</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">First response, n/N (%)</td><td align="left" valign="bottom">Modal response, n/N (%)</td><td align="left" valign="bottom">First response</td><td align="left" valign="bottom">Modal response</td></tr></thead><tbody><tr><td align="left" valign="top">GPT-4o</td><td align="left" valign="top">78/107 (72.9)</td><td align="left" valign="top">70/101 (69.3)</td><td align="left" valign="top">0.434</td><td align="left" valign="top">0.335</td></tr><tr><td align="left" valign="top">GPT-5.2</td><td align="left" valign="top">73/107 (68.2)</td><td align="left" valign="top">66/97 (68.0)</td><td align="left" valign="top">0.369</td><td align="left" valign="top">0.306</td></tr><tr><td align="left" valign="top">Gemini 3 Pro</td><td align="left" valign="top">72/107 (67.3)</td><td align="left" valign="top">76/102 (74.5)</td><td align="left" valign="top">0.286</td><td align="left" valign="top">0.434</td></tr><tr><td align="left" valign="top">Claude Sonnet 4.5</td><td align="left" valign="top">52/107 (48.6)</td><td align="left" valign="top">65/98 (66.3)</td><td align="left" valign="top">0.115</td><td align="left" valign="top">0.356</td></tr></tbody></table></table-wrap></sec><sec id="s3-5"><title>Predictors of Complete Discordance</title><p>Among the included cases, complete discordance&#x2014;defined as cases in which all evaluated LLMs generated treatment recommendations different from the MDT decision&#x2014;was observed in 17 of 107 (15.9%) cases. Complete discordance did not occur in any of the 31 unresectable cases, compared with 9 of 41 borderline resectable and 8 of 35 resectable cases (Fisher exact test, <italic>P</italic>=.003); because no events occurred in the unresectable stratum, resectability could not be entered into the regression model. To identify clinical factors associated with complete discordance, logistic regression analyses were performed (<xref ref-type="table" rid="table4">Table 4</xref>). In the univariate analysis, recurrent or on-treatment disease (OR 5.00, 95% CI 1.69&#x2010;14.82; <italic>P</italic>=.004), pancreatic tumor location (OR 3.98, 95% CI 1.35&#x2010;11.67; <italic>P</italic>=.01), and low MDT agreement level (OR 5.25, 95% CI 1.03&#x2010;26.66; <italic>P</italic>=.045) were significantly associated with complete discordance. In the multivariate analysis, recurrent or on-treatment disease (adjusted OR 5.40, 95% CI 1.65&#x2010;17.68; <italic>P</italic>=.005), pancreatic tumor location (adjusted OR 7.37, 95% CI 2.03&#x2010;26.78; <italic>P</italic>=.002), and low MDT agreement level (adjusted OR 10.33, 95% CI 1.54&#x2010;69.38; <italic>P</italic>=.02) remained independent predictors of complete discordance.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Univariate and multivariate analyses of predictors of complete discordance.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Variables</td><td align="left" valign="bottom">Crude OR<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup> (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td><td align="left" valign="bottom">Adjusted OR (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Sex (male vs female)</td><td align="left" valign="top">1.17 (0.40-3.44)</td><td align="left" valign="top">.77</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top">Age (per year)</td><td align="left" valign="top">0.98 (0.95-1.00)</td><td align="left" valign="top">.49</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top">Disease status: recurrent vs primary</td><td align="left" valign="top">5.00 (1.69-14.82)</td><td align="left" valign="top">.004</td><td align="left" valign="top">5.40 (1.65-17.68)</td><td align="left" valign="top">.005</td></tr><tr><td align="left" valign="top">Tumor location: pancreas vs biliary</td><td align="left" valign="top">3.98 (1.35-11.67)</td><td align="left" valign="top">.01</td><td align="left" valign="top">7.37 (2.03-26.78)</td><td align="left" valign="top">.002</td></tr><tr><td align="left" valign="top" colspan="5">Main department (reference: gastroenterology)</td></tr><tr><td align="left" valign="top">&#x2003;Medical oncology</td><td align="left" valign="top">4.24 (1.37-13.07)</td><td align="left" valign="top">.01</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top">&#x2003;Surgical oncology</td><td align="left" valign="top">0.47 (0.05-4.10)</td><td align="left" valign="top">.50</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top" colspan="5">MDT<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup> agreement (reference: high)</td></tr><tr><td align="left" valign="top">&#x2003;Moderate</td><td align="left" valign="top">2.33 (0.55-9.96)</td><td align="left" valign="top">.25</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top">&#x2003;Low</td><td align="left" valign="top">5.25 (1.03-26.66)</td><td align="left" valign="top">.045</td><td align="left" valign="top">10.33 (1.54-69.38)</td><td align="left" valign="top">.02</td></tr><tr><td align="left" valign="top">CEA<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup> (per ng/mL)</td><td align="left" valign="top">1.00 (1.00-1.00)</td><td align="left" valign="top">.93</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top">CA 19&#x2010;9<sup><xref ref-type="table-fn" rid="table4fn5">e</xref></sup> (per U/mL)</td><td align="left" valign="top">1.00 (1.00-1.00)</td><td align="left" valign="top">.66</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2003;</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>OR: odds ratio.</p></fn><fn id="table4fn2"><p><sup>b</sup>Not available.</p></fn><fn id="table4fn3"><p><sup>c</sup>MDT: multidisciplinary team.</p></fn><fn id="table4fn4"><p><sup>d</sup>CEA: carcinoembryonic antigen.</p></fn><fn id="table4fn5"><p><sup>e</sup>CA 19&#x2010;9: carbohydrate antigen 19-9.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>HPB malignancies require complex treatment planning that integrates radiologic findings, pathology, surgical considerations, and systemic therapy strategies [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. Consequently, MDT discussions have become a cornerstone of modern oncologic care, with previous studies demonstrating that MDT-based management can improve diagnostic accuracy, treatment selection, and, in some settings, patient survival outcomes [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref5">5</xref>-<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref21">21</xref>]. However, MDT decision-making also requires substantial coordination of heterogeneous clinical information and expert perspectives, making the process resource-intensive and difficult to scale across health care systems [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref22">22</xref>].</p><p>Recent advances in LLMs have raised interest in their potential role in supporting complex clinical reasoning [<xref ref-type="bibr" rid="ref9">9</xref>-<xref ref-type="bibr" rid="ref13">13</xref>]. Most prior studies, however, have primarily evaluated the accuracy or guideline concordance of LLM-generated recommendations rather than examining how these systems behave within real-world multidisciplinary decision environments [<xref ref-type="bibr" rid="ref11">11</xref>-<xref ref-type="bibr" rid="ref14">14</xref>]. In this study, using real-world HPB MDT cases, we evaluated the feasibility of LLMs operating within an MDT-like clinical decision context [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref14">14</xref>]. Our results show that, while LLM-generated recommendations demonstrated moderate alignment with MDT decisions, their outputs exhibited model-dependent variability and diverged most frequently in clinically complex scenarios such as pancreatic malignancies and cases with low MDT consensus.</p><p>An important observation from this study relates to the stability of LLM-generated treatment recommendations. In clinical decision-making, the consistency of recommendations is essential because clinicians expect similar conclusions when the same clinical information is presented. However, LLMs generate responses through probabilistic processes, which may lead to variability even when identical inputs are provided. Despite the increasing interest in applying LLMs to clinical decision support, the stability of model-generated recommendations in real-world clinical scenarios has rarely been examined. By repeatedly querying multiple LLMs with identical MDT case summaries, our study demonstrates that stability varies substantially across models. These findings highlight that response stability represents an important consideration when integrating LLM-based systems into clinical workflows and suggest that some models may be more suitable than others for supporting complex multidisciplinary decision-making.</p></sec><sec id="s4-2"><title>Concordance and Its Interpretation</title><p>In our study, we observed a moderate level of concordance between LLM-generated treatment recommendations and MDT decisions, with overall agreement rates generally ranging between approximately 60% and 70% across models. Although MDT decisions are widely regarded as the current standard for complex oncologic decision-making, they do not necessarily represent a single definitive &#x201C;correct&#x201D; answer. Rather, MDT recommendations emerge from multidisciplinary discussions that integrate clinical expertise, institutional practice patterns, and patient-specific considerations. Consequently, some degree of variation in treatment recommendations can be expected even among expert clinicians.</p><p>Within this context, the concordance observed in our study suggests that LLM-generated recommendations may capture certain elements of multidisciplinary clinical reasoning. Discordant cases may, therefore, not necessarily represent incorrect recommendations but instead reflect institutional treatment practices, patient-specific considerations, or contextual factors that may not be fully captured in structured case summaries. Taken together, these findings indicate that LLM recommendations demonstrate a meaningful degree of alignment with MDT-based clinical decision-making.</p><p>Three observations qualify these concordance estimates. First, chemotherapy accounted for 77 of the 107 (72.0%) MDT decisions, so overall concordance is heavily influenced by the ability to identify this dominant category. Class-wise <italic>F</italic><sub>1</sub>-scores were consistently lower for surgery than for chemotherapy across all models, indicating that the models were least reliable for the decision carrying the greatest clinical consequence. Second, concordance estimates depended on which of the repeated responses was used as the reference, and the ranking of models differed between the first response and modal response definitions. We therefore report both rather than designating one as correct: the first response reflects what a clinician would receive from a single query, whereas the modal response reflects the model&#x2019;s central tendency, and the two answer different questions. That they diverge is itself a manifestation of response instability and supports our central point that concordance and stability should not be interpreted independently. Third, treatment recommendations were categorized by the modality initiated first, so a model recommendation of neoadjuvant chemotherapy followed by surgery and an MDT decision of upfront surgery were counted as discordant even though both anticipate resection; concordance as reported, therefore, reflects agreement on the immediate treatment modality rather than on the complete therapeutic strategy (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p><p>Two further considerations bear on how these estimates should be used. Concordance figures reported in other tumor board series, including approximately 80% for cholangiocarcinoma in a single-center liver MDT study [<xref ref-type="bibr" rid="ref19">19</xref>], are not directly comparable to ours: that design used curated case summaries and a coarser outcome definition, whereas our outcome had 5 categories and our cases were consecutive real presentations, both of which lower concordance mechanically. More importantly, the rationales accompanying model recommendations were fluent and clinically framed, regardless of whether the recommendation was correct (<xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>), consistent with evidence that generated reasoning does not reliably reflect the computation that produces the answer [<xref ref-type="bibr" rid="ref23">23</xref>]. A clinician reviewing an LLM recommendation, therefore, cannot treat the persuasiveness of its justification as a quality signal, and verification requires checking each stated premise against the source reports rather than assessing the coherence of the explanation.</p></sec><sec id="s4-3"><title>Predictors of Discordance</title><p>Discordance between LLM-generated recommendations and MDT decisions in our study was not randomly distributed but occurred more frequently in specific clinical contexts, particularly in pancreatic tumors and cases with low MDT agreement. While the association with low MDT confidence is expected, the higher discordance observed in pancreatic malignancies warrants further consideration. Treatment planning for pancreatic cancer often requires nuanced interpretation of tumor resectability, vascular involvement, and patient-specific clinical factors, which can introduce variability in clinical decision-making.</p><p>Previous studies have also shown that substantial variation may exist among expert teams in the assessment of pancreatic cancer resectability and treatment strategies [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]. In multicenter evaluations of pancreatic cancer cases reviewed by multiple MDTs, notable differences in resectability assessments and treatment recommendations have been reported [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]. These findings suggest that pancreatic cancer represents a clinical scenario in which decision-making itself is inherently complex and subject to differing expert interpretations [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]. In this context, the discordance observed between LLM recommendations and MDT decisions in our study may reflect the intrinsic difficulty of these cases rather than a systematic limitation of model reasoning.</p></sec><sec id="s4-4"><title>Clinical Implications</title><p>Our findings highlight a potential and pragmatic role for LLMs in multidisciplinary clinical decision environments. Rather than representing a fully autonomous clinical decision system, LLM-based approaches may function as reasoning-support tools that assist clinicians in organizing complex clinical information and generating structured treatment considerations. In this sense, the use of LLMs within MDT-like contexts may represent a feasible direction for further investigation.</p><p>This study evaluated retrospective case summaries in a controlled setting. It did not assess safety, clinician interaction, workflow impact, or patient outcomes, and the implications discussed here are, therefore, framed as conditions that would have to be met before any clinical use, not as evidence that such use is warranted. Any such use would require safeguards consistent with the limitations observed here. Because model outputs varied across identical queries and were least accurate for surgical decisions, every recommendation would require review by the treating team, which retains clinical accountability. Deployment would also require version pinning with periodic revalidation, given that model behavior changes with each release, as well as the processing of only deidentified text.</p><p>At the same time, the application of AI to complex oncologic decision-making will likely require the integration of additional components beyond language-based reasoning. Comprehensive clinical decision systems would need to incorporate multimodal capabilities, including agents capable of directly interpreting imaging data, pathology findings, and other raw clinical inputs. Our findings indicate that the reasoning component of such systems, represented by LLM-based synthesis of clinical information and treatment considerations, remains limited in both stability and agreement with expert consensus and that these properties would need to be established before such a component could be relied upon.</p></sec><sec id="s4-5"><title>Limitations</title><p>This study has several limitations that should be considered when interpreting the findings. First, the analysis was conducted using cases from a single-center MDT conference, which may limit the generalizability of the results to other institutions with different clinical practices or MDT structures. Second, the clinical information provided to the LLMs consisted of structured textual summaries rather than raw clinical data. In real-world clinical environments, decision-making may also involve direct interpretation of imaging studies, pathology slides, and other multimodal data that were not incorporated in the present study. Third, the number of analyzed cases was relatively limited, and larger datasets may be necessary to further validate the observed patterns of concordance and discordance. Fourth, the performance characteristics of LLMs evolve rapidly over time as new model versions are released; therefore, the results of this study should be interpreted within the context of the specific model versions evaluated. In addition, the models were queried in separate blocks over a 7-month period rather than concurrently, and each was evaluated in the version and interface configuration current at that time. Because consumer interfaces do not expose decoding parameters and, for GPT-5.2, route queries between variants without user control, exact reproduction of these outputs is not possible; this constrains reproducibility but reflects the conditions under which clinicians would actually access these models. Finally, concordance estimates depended on how the reference response was defined among repeated queries. We report both the first and the modal response, but neither fully resolves which output should be regarded as a model&#x2019;s recommendation, and defining this reference remains an open methodological problem in the evaluation of generative models. Complete discordance was defined using the first response of each model and was not recalculated under the modal definition. The multivariate analysis was based on 17 events and is exploratory; the estimate for low MDT agreement derives from 7 cases and should be interpreted with caution.</p></sec><sec id="s4-6"><title>Conclusions</title><p>Our study demonstrates that LLMs can generate treatment recommendations that show a meaningful degree of alignment with MDT-based decision-making in HPB oncology. Importantly, response stability varied substantially across models, with discordance rates ranging from 12.8% to 30.2%, indicating that concordance alone is insufficient for evaluating LLMs as clinical decision support tools. Although variability in model responses and discordance in certain clinical contexts were observed, these findings suggest that LLM-based systems may feasibly support certain aspects of multidisciplinary clinical reasoning. Before LLMs are integrated into clinical workflows, their response stability must be systematically characterized alongside concordance. As AI technologies continue to evolve, further research integrating multimodal clinical data and multidisciplinary expertise will be essential to better define the role of LLMs within real-world oncologic decision-making processes.</p></sec></sec></body><back><ack><p>Generative AI tools were used during the preparation of this manuscript and are disclosed in accordance with JMIR Publications' policy. Claude (Anthropic) was used for the generation of the figures, for preliminary verification of the statistical analyses, and for English-language editing. The statistical results reported in the manuscript were computed by the authors in R and are reported from that analysis; all text was reviewed and approved by the authors. No generative AI tool is listed as an author, and the authors accept full responsibility for the content of the manuscript. Separately, and as the object of study rather than as a manuscript preparation tool, the free-text rationales reproduced in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref> are outputs of Gemini 3 Pro, generated as study data.</p></ack><notes><sec><title>Funding</title><p>This work was supported by National Research Foundation of Korea (NRF) grants funded by the Korean government (Ministry of Science and ICT) (RS-2023-NR076871 and RS-2025-02218795), and by the Korea Health Industry Development Institute (KHIDI), funded by the Ministry of Health and Welfare, Republic of Korea (RS-2025-02223415).</p></sec><sec><title>Data Availability</title><p>Data cannot be shared publicly for legal reasons. Data are available from the Bundang CHA Medical Center Institutional Data Access/Ethics Committee (contact via sungjun88@chamc.co.kr) for researchers who meet the criteria for access to confidential data.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: SJJ, SHL</p><p>Data curation: SJJ</p><p>Formal analysis: SJJ</p><p>Methodology: SJJ</p><p>Project administration: SJY, SHL</p><p>Resources: EHC, IK, SJY, BK, JSK, HJC, CK, MJS, SPS, CA, HYL, JSY, SJ, JHI, KHK</p><p>Supervision: SHL</p><p>Visualization: SJJ</p><p>Writing &#x2013; original draft: SJJ</p><p>Writing &#x2013; review &#x0026; editing: EHC, IK, SJY, BK, JSK, HJC, CK, MJS, SPS, CA, HYL, JSY, SJ, JHI, KHK, SHL</p><p>All authors read and approved the final manuscript.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">HPB</term><def><p>hepatopancreatobiliary</p></def></def-item><def-item><term id="abb2">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb3">MDT</term><def><p>multidisciplinary team</p></def></def-item><def-item><term id="abb4">OR</term><def><p>odds ratio</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yao</surname><given-names>W</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>X</given-names> </name><name name-style="western"><surname>Fan</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Multidisciplinary team diagnosis and treatment of pancreatic cancer: current landscape and future prospects</article-title><source>Front Oncol</source><year>2023</year><volume>13</volume><fpage>1077605</fpage><pub-id pub-id-type="doi">10.3389/fonc.2023.1077605</pub-id><pub-id pub-id-type="medline">37007078</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pawlik</surname><given-names>TM</given-names> </name><name name-style="western"><surname>Laheru</surname><given-names>D</given-names> </name><name name-style="western"><surname>Hruban</surname><given-names>RH</given-names> </name><etal/></person-group><article-title>Evaluating the impact of a single-day multidisciplinary clinic on the management of pancreatic cancer</article-title><source>Ann Surg Oncol</source><year>2008</year><month>08</month><volume>15</volume><issue>8</issue><fpage>2081</fpage><lpage>2088</lpage><pub-id pub-id-type="doi">10.1245/s10434-008-9929-7</pub-id><pub-id pub-id-type="medline">18461404</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Quero</surname><given-names>G</given-names> </name><name name-style="western"><surname>De Sio</surname><given-names>D</given-names> </name><name name-style="western"><surname>Fiorillo</surname><given-names>C</given-names> </name><etal/></person-group><article-title>The role of the multidisciplinary tumor board (MDTB) in the assessment of pancreatic cancer diagnosis and resectability: a tertiary referral center experience</article-title><source>Front Surg</source><year>2023</year><volume>10</volume><fpage>1119557</fpage><pub-id pub-id-type="doi">10.3389/fsurg.2023.1119557</pub-id><pub-id pub-id-type="medline">36874464</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Casadio</surname><given-names>M</given-names> </name><name name-style="western"><surname>Cardinale</surname><given-names>V</given-names> </name><name name-style="western"><surname>Kl&#x00FC;mpen</surname><given-names>HJ</given-names> </name><etal/></person-group><article-title>Setup of multidisciplinary team discussions for patients with cholangiocarcinoma: current practice and recommendations from the European Network for the Study of Cholangiocarcinoma (ENS-CCA)</article-title><source>ESMO Open</source><year>2022</year><month>02</month><volume>7</volume><issue>1</issue><fpage>100377</fpage><pub-id pub-id-type="doi">10.1016/j.esmoop.2021.100377</pub-id><pub-id pub-id-type="medline">35093741</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Freeman</surname><given-names>RK</given-names> </name><name name-style="western"><surname>Van Woerkom</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Vyverberg</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ascioti</surname><given-names>AJ</given-names> </name></person-group><article-title>The effect of a multidisciplinary thoracic malignancy conference on the treatment of patients with esophageal cancer</article-title><source>Ann Thorac Surg</source><year>2011</year><month>10</month><volume>92</volume><issue>4</issue><fpage>1239</fpage><lpage>1242</lpage><pub-id pub-id-type="doi">10.1016/j.athoracsur.2011.05.057</pub-id><pub-id pub-id-type="medline">21867990</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kesson</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Allardice</surname><given-names>GM</given-names> </name><name name-style="western"><surname>George</surname><given-names>WD</given-names> </name><name name-style="western"><surname>Burns</surname><given-names>HJG</given-names> </name><name name-style="western"><surname>Morrison</surname><given-names>DS</given-names> </name></person-group><article-title>Effects of multidisciplinary team working on breast cancer survival: retrospective, comparative, interventional cohort study of 13 722 women</article-title><source>BMJ</source><year>2012</year><month>04</month><day>26</day><volume>344</volume><fpage>e2718</fpage><pub-id pub-id-type="doi">10.1136/bmj.e2718</pub-id><pub-id pub-id-type="medline">22539013</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pillay</surname><given-names>B</given-names> </name><name name-style="western"><surname>Wootten</surname><given-names>AC</given-names> </name><name name-style="western"><surname>Crowe</surname><given-names>H</given-names> </name><etal/></person-group><article-title>The impact of multidisciplinary team meetings on patient assessment, management and outcomes in oncology settings: a systematic review of the literature</article-title><source>Cancer Treat Rev</source><year>2016</year><month>01</month><volume>42</volume><fpage>56</fpage><lpage>72</lpage><pub-id pub-id-type="doi">10.1016/j.ctrv.2015.11.007</pub-id><pub-id pub-id-type="medline">26643552</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Brauer</surname><given-names>DG</given-names> </name><name name-style="western"><surname>Strand</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Sanford</surname><given-names>DE</given-names> </name><etal/></person-group><article-title>Utility of a multidisciplinary tumor board in the management of pancreatic and upper gastrointestinal diseases: an observational study</article-title><source>HPB (Oxford)</source><year>2017</year><month>02</month><volume>19</volume><issue>2</issue><fpage>133</fpage><lpage>139</lpage><pub-id pub-id-type="doi">10.1016/j.hpb.2016.11.002</pub-id><pub-id pub-id-type="medline">27916436</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moor</surname><given-names>M</given-names> </name><name name-style="western"><surname>Banerjee</surname><given-names>O</given-names> </name><name name-style="western"><surname>Abad</surname><given-names>ZSH</given-names> </name><etal/></person-group><article-title>Foundation models for generalist medical artificial intelligence</article-title><source>Nature</source><year>2023</year><month>04</month><volume>616</volume><issue>7956</issue><fpage>259</fpage><lpage>265</lpage><pub-id pub-id-type="doi">10.1038/s41586-023-05881-4</pub-id><pub-id pub-id-type="medline">37045921</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Benary</surname><given-names>M</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>XD</given-names> </name><name name-style="western"><surname>Schmidt</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Leveraging large language models for decision support in personalized oncology</article-title><source>JAMA Netw Open</source><year>2023</year><month>11</month><day>1</day><volume>6</volume><issue>11</issue><fpage>e2343689</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2023.43689</pub-id><pub-id pub-id-type="medline">37976064</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kaiser</surname><given-names>KN</given-names> </name><name name-style="western"><surname>Hughes</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>AD</given-names> </name><etal/></person-group><article-title>Use of large language models as clinical decision support tools for management of pancreatic adenocarcinoma using National Comprehensive Cancer Network guidelines</article-title><source>Surgery</source><year>2025</year><month>06</month><volume>182</volume><fpage>109267</fpage><pub-id pub-id-type="doi">10.1016/j.surg.2025.109267</pub-id><pub-id pub-id-type="medline">40055080</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thirunavukarasu</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSJ</given-names> </name><name name-style="western"><surname>Elangovan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Gutierrez</surname><given-names>L</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>TF</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSW</given-names> </name></person-group><article-title>Large language models in medicine</article-title><source>Nat Med</source><year>2023</year><month>08</month><volume>29</volume><issue>8</issue><fpage>1930</fpage><lpage>1940</lpage><pub-id pub-id-type="doi">10.1038/s41591-023-02448-8</pub-id><pub-id pub-id-type="medline">37460753</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ah-Thiane</surname><given-names>L</given-names> </name><name name-style="western"><surname>Heudel</surname><given-names>PE</given-names> </name><name name-style="western"><surname>Campone</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Large language models as decision-making tools in oncology: comparing artificial intelligence suggestions and expert recommendations</article-title><source>JCO Clin Cancer Inform</source><year>2025</year><month>03</month><volume>9</volume><fpage>e2400230</fpage><pub-id pub-id-type="doi">10.1200/CCI-24-00230</pub-id><pub-id pub-id-type="medline">40112233</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nardone</surname><given-names>V</given-names> </name><name name-style="western"><surname>Marmorino</surname><given-names>F</given-names> </name><name name-style="western"><surname>Germani</surname><given-names>MM</given-names> </name><etal/></person-group><article-title>The role of artificial intelligence on tumor boards: perspectives from surgeons, medical oncologists and radiation oncologists</article-title><source>Curr Oncol</source><year>2024</year><month>08</month><day>27</day><volume>31</volume><issue>9</issue><fpage>4984</fpage><lpage>5007</lpage><pub-id pub-id-type="doi">10.3390/curroncol31090369</pub-id><pub-id pub-id-type="medline">39329997</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>D</given-names> </name><name name-style="western"><surname>Alnassar</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Avison</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>RS</given-names> </name><name name-style="western"><surname>Raman</surname><given-names>S</given-names> </name></person-group><article-title>Large language model applications for health information extraction in oncology: scoping review</article-title><source>JMIR Cancer</source><year>2025</year><month>03</month><day>28</day><volume>11</volume><fpage>e65984</fpage><pub-id pub-id-type="doi">10.2196/65984</pub-id><pub-id pub-id-type="medline">40153782</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wong</surname><given-names>A</given-names> </name><name name-style="western"><surname>Otles</surname><given-names>E</given-names> </name><name name-style="western"><surname>Donnelly</surname><given-names>JP</given-names> </name><etal/></person-group><article-title>External validation of a widely implemented proprietary sepsis prediction model in hospitalized patients</article-title><source>JAMA Intern Med</source><year>2021</year><month>08</month><day>1</day><volume>181</volume><issue>8</issue><fpage>1065</fpage><lpage>1070</lpage><pub-id pub-id-type="doi">10.1001/jamainternmed.2021.2626</pub-id><pub-id pub-id-type="medline">34152373</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Habib</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>AL</given-names> </name><name name-style="western"><surname>Grant</surname><given-names>RW</given-names> </name></person-group><article-title>The Epic Sepsis Model falls short&#x2014;the importance of external validation</article-title><source>JAMA Intern Med</source><year>2021</year><month>08</month><day>1</day><volume>181</volume><issue>8</issue><fpage>1040</fpage><lpage>1041</lpage><pub-id pub-id-type="doi">10.1001/jamainternmed.2021.3333</pub-id><pub-id pub-id-type="medline">34152360</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Griewing</surname><given-names>S</given-names> </name><name name-style="western"><surname>Knitza</surname><given-names>J</given-names> </name><name name-style="western"><surname>Boekhoff</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Evolution of publicly available large language models for complex decision-making in breast cancer care</article-title><source>Arch Gynecol Obstet</source><year>2024</year><month>07</month><volume>310</volume><issue>1</issue><fpage>537</fpage><lpage>550</lpage><pub-id pub-id-type="doi">10.1007/s00404-024-07565-4</pub-id><pub-id pub-id-type="medline">38806945</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Palzer</surname><given-names>J</given-names> </name><name name-style="western"><surname>Genchev</surname><given-names>A</given-names> </name><name name-style="western"><surname>Belger</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Large language models for multidisciplinary tumor board decision-making in primary liver tumors: a retrospective single-center study</article-title><source>Cancers (Basel)</source><year>2026</year><month>07</month><day>7</day><volume>18</volume><issue>13</issue><fpage>2175</fpage><pub-id pub-id-type="doi">10.3390/cancers18132175</pub-id><pub-id pub-id-type="medline">42449717</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gallifant</surname><given-names>J</given-names> </name><name name-style="western"><surname>Afshar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ameen</surname><given-names>S</given-names> </name><etal/></person-group><article-title>The TRIPOD-LLM reporting guideline for studies using large language models</article-title><source>Nat Med</source><year>2025</year><month>01</month><volume>31</volume><issue>1</issue><fpage>60</fpage><lpage>69</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03425-5</pub-id><pub-id pub-id-type="medline">39779929</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Quero</surname><given-names>G</given-names> </name><name name-style="western"><surname>Salvatore</surname><given-names>L</given-names> </name><name name-style="western"><surname>Fiorillo</surname><given-names>C</given-names> </name><etal/></person-group><article-title>The impact of the multidisciplinary tumor board (MDTB) on the management of pancreatic diseases in a tertiary referral center</article-title><source>ESMO Open</source><year>2021</year><month>02</month><volume>6</volume><issue>1</issue><fpage>100010</fpage><pub-id pub-id-type="doi">10.1016/j.esmoop.2020.100010</pub-id><pub-id pub-id-type="medline">33399076</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Horvitz</surname><given-names>E</given-names> </name><name name-style="western"><surname>Mulligan</surname><given-names>D</given-names> </name></person-group><article-title>Policy forum. Data, privacy, and the greater good</article-title><source>Science</source><year>2015</year><month>07</month><day>17</day><volume>349</volume><issue>6245</issue><fpage>253</fpage><lpage>255</lpage><pub-id pub-id-type="doi">10.1126/science.aac4520</pub-id><pub-id pub-id-type="medline">26185242</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Turpin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Michael</surname><given-names>J</given-names> </name><name name-style="western"><surname>Perez</surname><given-names>E</given-names> </name><name name-style="western"><surname>Bowman</surname><given-names>SR</given-names> </name></person-group><article-title>Language models don&#x2019;t always say what they think: unfaithful explanations in chain-of-thought prompting</article-title><source>Adv Neural Inf Process Syst</source><year>2023</year><fpage>74952</fpage><lpage>74965</lpage><pub-id pub-id-type="doi">10.52202/075280-3275</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bockhorn</surname><given-names>M</given-names> </name><name name-style="western"><surname>Uzunoglu</surname><given-names>FG</given-names> </name><name name-style="western"><surname>Adham</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Borderline resectable pancreatic cancer: a consensus statement by the International Study Group of Pancreatic Surgery (ISGPS)</article-title><source>Surgery</source><year>2014</year><month>06</month><volume>155</volume><issue>6</issue><fpage>977</fpage><lpage>988</lpage><pub-id pub-id-type="doi">10.1016/j.surg.2014.02.001</pub-id><pub-id pub-id-type="medline">24856119</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wittel</surname><given-names>UA</given-names> </name><name name-style="western"><surname>Lubgan</surname><given-names>D</given-names> </name><name name-style="western"><surname>Ghadimi</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Consensus in determining the resectability of locally progressed pancreatic ductal adenocarcinoma - results of the CONKO-007 multicenter trial</article-title><source>BMC Cancer</source><year>2019</year><month>10</month><day>22</day><volume>19</volume><issue>1</issue><fpage>979</fpage><pub-id pub-id-type="doi">10.1186/s12885-019-6148-5</pub-id><pub-id pub-id-type="medline">31640628</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Treatment category mapping rules and institutional protocols.</p><media xlink:href="jmir_v28i1e101137_app1.docx" xlink:title="DOCX File, 9 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Annotated model rationales for 4 representative cases.</p><media xlink:href="jmir_v28i1e101137_app2.docx" xlink:title="DOCX File, 18 KB"/></supplementary-material></app-group></back></article>