<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="2.0">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">JMIR</journal-id>
      <journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id>
      <journal-title>Journal of Medical Internet Research</journal-title>
      <issn pub-type="epub">1438-8871</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v28i1e92407</article-id>
      <article-id pub-id-type="pmid">42627683</article-id>
      <article-id pub-id-type="doi">10.2196/92407</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Original Paper</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Original Paper</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Zero-Shot Classification of Postoperative Complications From Real-World Discharge Letters According to the Clavien-Dindo System Using Large Language Models in Liver Surgery: Comparative Study</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>Steenstra</surname>
            <given-names>Ivan</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Chrimes</surname>
            <given-names>Dillon</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Li</surname>
            <given-names>Zhi</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Warmer</surname>
            <given-names>Sina</given-names>
          </name>
          <degrees>MSc</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <address>
            <institution>Institute for Artificial Intelligence in Medicine (IKIM)</institution>
            <institution>University Hospital Essen</institution>
            <addr-line>Hufelandstraße 55</addr-line>
            <addr-line>Essen, 45147</addr-line>
            <country>Germany</country>
            <phone>1 49 201 723 77816</phone>
            <email>sina.warmer@uk-essen.de</email>
          </address>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0002-2262-2655</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author">
          <name name-style="western">
            <surname>Arzideh</surname>
            <given-names>Kamyar</given-names>
          </name>
          <degrees>MSc</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0005-6074-804X</ext-link>
        </contrib>
        <contrib id="contrib3" contrib-type="author">
          <name name-style="western">
            <surname>Morys</surname>
            <given-names>Marie</given-names>
          </name>
          <xref rid="aff1" ref-type="aff">1</xref>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0006-4528-2322</ext-link>
        </contrib>
        <contrib id="contrib4" contrib-type="author">
          <name name-style="western">
            <surname>Idrissi-Yaghir</surname>
            <given-names>Ahmad</given-names>
          </name>
          <degrees>MSc</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-1507-9690</ext-link>
        </contrib>
        <contrib id="contrib5" contrib-type="author">
          <name name-style="western">
            <surname>Bednarsch</surname>
            <given-names>Jan</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff3" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-8143-6452</ext-link>
        </contrib>
        <contrib id="contrib6" contrib-type="author">
          <name name-style="western">
            <surname>Heise</surname>
            <given-names>Daniel</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff3" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-6923-0849</ext-link>
        </contrib>
        <contrib id="contrib7" contrib-type="author">
          <name name-style="western">
            <surname>Oezcelik</surname>
            <given-names>Arzu</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff3" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-8353-9532</ext-link>
        </contrib>
        <contrib id="contrib8" contrib-type="author">
          <name name-style="western">
            <surname>Reschke</surname>
            <given-names>Marc</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff3" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0002-3113-4741</ext-link>
        </contrib>
        <contrib id="contrib9" contrib-type="author">
          <name name-style="western">
            <surname>Ulmer</surname>
            <given-names>Tom F</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff3" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0006-5793-1959</ext-link>
        </contrib>
        <contrib id="contrib10" contrib-type="author">
          <name name-style="western">
            <surname>Neumann</surname>
            <given-names>Ulf P</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff3" ref-type="aff">3</xref>
          <xref rid="aff4" ref-type="aff">4</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-3831-8917</ext-link>
        </contrib>
        <contrib id="contrib11" contrib-type="author">
          <name name-style="western">
            <surname>Nensa</surname>
            <given-names>Felix</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-5811-7100</ext-link>
        </contrib>
        <contrib id="contrib12" contrib-type="author">
          <name name-style="western">
            <surname>Borys</surname>
            <given-names>Katarzyna</given-names>
          </name>
          <degrees>MSc</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-6987-6041</ext-link>
        </contrib>
        <contrib id="contrib13" contrib-type="author" equal-contrib="yes">
          <name name-style="western">
            <surname>Hosch</surname>
            <given-names>René</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-1760-2342</ext-link>
        </contrib>
        <contrib id="contrib14" contrib-type="author" equal-contrib="yes">
          <name name-style="western">
            <surname>Schmitz</surname>
            <given-names>Sophia M</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff3" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-6732-1595</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff1">
        <label>1</label>
        <institution>Institute for Artificial Intelligence in Medicine (IKIM)</institution>
        <institution>University Hospital Essen</institution>
        <addr-line>Essen</addr-line>
        <country>Germany</country>
      </aff>
      <aff id="aff2">
        <label>2</label>
        <institution>Institute of Diagnostic and Interventional Radiology and Neuroradiology</institution>
        <institution>University Hospital Essen</institution>
        <addr-line>Essen</addr-line>
        <country>Germany</country>
      </aff>
      <aff id="aff3">
        <label>3</label>
        <institution>Department of General, Visceral, Vascular and Transplantation Surgery</institution>
        <institution>University Hospital Essen</institution>
        <addr-line>Essen</addr-line>
        <country>Germany</country>
      </aff>
      <aff id="aff4">
        <label>4</label>
        <institution>Department of Surgery</institution>
        <institution>Maastricht UMC+</institution>
        <addr-line>Maastricht</addr-line>
        <country>The Netherlands</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Sina Warmer <email>sina.warmer@uk-essen.de</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>21</day>
        <month>8</month>
        <year>2026</year>
      </pub-date>
      <volume>28</volume>
      <elocation-id>e92407</elocation-id>
      <history>
        <date date-type="received">
          <day>29</day>
          <month>1</month>
          <year>2026</year>
        </date>
        <date date-type="rev-request">
          <day>8</day>
          <month>5</month>
          <year>2026</year>
        </date>
        <date date-type="rev-recd">
          <day>16</day>
          <month>7</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>16</day>
          <month>7</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©Sina Warmer, Kamyar Arzideh, Marie Morys, Ahmad Idrissi-Yaghir, Jan Bednarsch, Daniel Heise, Arzu Oezcelik, Marc Reschke, Tom F Ulmer, Ulf P Neumann, Felix Nensa, Katarzyna Borys, René Hosch, Sophia M Schmitz. Originally published in the Journal of Medical Internet Research (https://www.jmir.org), 21.08.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on https://www.jmir.org/, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://www.jmir.org/2026/1/e92407" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>The standardized extraction of postoperative complications from unstructured routine clinical documentation remains a major unresolved challenge in digital surgery and health informatics. Although the Clavien-Dindo classification is the established standard for grading postoperative complications, its application in routine clinical documentation is largely implicit and unstructured, limiting scalable quality assessment in surgical care.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>This study aimed to assess the capability of open-weight and proprietary large language models (LLMs) to classify postoperative complications according to the Clavien-Dindo system using discharge letters, benchmarked against expert annotation.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>We analyzed discharge letters from 650 surgical cases of 649 patients (median 67, IQR 58-73 y; 229/649, 35% female) who underwent hepatobiliary surgery between 2010 and 2024. The cohort included grade I-II complications in 24% (153/650), grade III-IV in 19% (121/650), and grade V (death) in 6% (42/650) of patients. A total of 4 open-weight (Qwen3-235B [Alibaba Cloud], Llama-3.3-70B [Meta AI], GPT-OSS-120B [OpenAI], Ministral-3-8B [Mistral AI]) and 2 proprietary (GPT 5.1 [OpenAI], Gemini 3 Pro [Google]) LLMs were prompted to infer complication grades directly from the discharge letters in a zero-shot setting. Model performance was evaluated against expert assessment using accuracy, <italic>F</italic><sub>1</sub>-scores, and Cohen κ. To assess interrater reliability and establish a human benchmark, a stratified 10% (n=65) subset was independently annotated by a second clinician, and Cohen κ was calculated between annotators and between each model and the primary expert.</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>Interrater agreement between the 2 clinical annotators yielded a Cohen κ of 0.75, providing a human benchmark for model performance interpretation. On the full 650-case dataset, open-weight models achieved accuracies ranging from 0.75 to 0.78 for fine-grained prediction, with weighted <italic>F</italic><sub>1</sub>-scores of 0.76-0.78 and macroaveraged <italic>F</italic><sub>1</sub>-scores of 0.50-0.63. For binary classification, accuracies ranged from 0.93 to 0.94, with weighted <italic>F</italic><sub>1</sub>-scores of 0.93-0.95 and Cohen κ of 0.76-0.79, approaching the human interrater benchmark. On a balanced 50-case subset, used as the sole basis for direct cross-model comparison, proprietary models achieved accuracies of 0.78 for fine-grained and 0.94-0.98 for binary classification. An ensemble approach yielded additional gains in classification performance.</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>LLMs demonstrated promising accuracy in classifying postoperative complications from discharge letters in a zero-shot setting, with performance approaching the upper bound of human interrater agreement. Open-weight models offer a particularly attractive trade-off between accuracy and computational efficiency, while ensemble strategies further enhance robustness. These results support the potential of LLMs to standardize complication assessment at scale and enable data-driven quality monitoring in surgical care.</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>large language models</kwd>
        <kwd>Clavien-Dindo classification</kwd>
        <kwd>AI in health care</kwd>
        <kwd>postoperative complications</kwd>
        <kwd>natural language processing</kwd>
        <kwd>health care interoperability</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <p>The automated extraction of clinically meaningful information from routine medical unstructured documentation has become a central ambition of digital medicine and health informatics [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref4">4</xref>]. The standardized classification of postoperative complications represents a particularly relevant and challenging benchmark. Over recent decades, the Clavien‑Dindo classification has emerged as the most widely used system for stratifying surgical complications by severity [<xref ref-type="bibr" rid="ref5">5</xref>]. Originally proposed by Pierre‑Alain Clavien and Daniel Dindo in 2004 [<xref ref-type="bibr" rid="ref6">6</xref>], the system grades complications based on their therapeutic consequences from grade I (any deviation from the normal postoperative course without need for pharmacologic, endoscopic, or radiologic intervention) through grade V (death of the patient). Its consistent application is important for surgical quality assurance, benchmarking, and outcome research. Additionally, its usage in randomized controlled surgical trials has strongly increased in the last decades [<xref ref-type="bibr" rid="ref5">5</xref>]. Yet, in everyday clinical workflows, complications are often recorded implicitly within unstructured text, such as operative notes, discharge letters, or nursing reports. This lack of structured representation hinders large-scale outcome monitoring and perpetuates bias through manual abstraction or selective reporting.</p>
      <p>Large language models (LLMs) have recently emerged as powerful tools for reasoning over unstructured clinical narratives, offering the potential to bridge this gap [<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref10">10</xref>]. These models demonstrate promising performance in natural language understanding, contextual inference, and classification [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref12">12</xref>], all without the need for manual feature engineering. However, the clinical adoption of LLMs must be considered in the context of data privacy, transparency, reproducibility, and interoperability. The growing ecosystem of open-weight and proprietary language models offers distinct trade-offs across these dimensions. Proprietary frontier models (eg, GPT-5 [OpenAI], Claude [Anthropic], Gemini [Google]) often achieve superior general reasoning but remain opaque and unsuitable for sensitive medical data due to closed architectures and uncertain data provenance [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref14">14</xref>]. In contrast, open-weight models (eg, Llama 3 by Meta AI [<xref ref-type="bibr" rid="ref15">15</xref>], Mistral by Mistral AI [<xref ref-type="bibr" rid="ref16">16</xref>], Qwen by Alibaba Cloud [<xref ref-type="bibr" rid="ref17">17</xref>], GPT-OSS by OpenAI [<xref ref-type="bibr" rid="ref18">18</xref>]) support local deployment and fine-tuning, enabling compliance with data protection regulations such as GDPR (General Data Protection Regulation) and the broader requirements of the European Union AI Act, while fostering reproducible, auditable research [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref21">21</xref>]. Small language models (SLMs) and domain-specific variants further enable resource-efficient deployment within hospital infrastructures [<xref ref-type="bibr" rid="ref22">22</xref>].</p>
      <p>In this study, we systematically evaluate and compare open-weight and proprietary LLMs for automatic Clavien-Dindo classification using only existing discharge letters from routine clinical documentation. Leveraging real-world surgical data, we examine each model’s ability to infer complication grades from unstructured text and quantify accuracy, consistency, and explainability. Moreover, we introduce a downstream interoperability layer that maps model outputs to Fast Healthcare Interoperability Resources (FHIR) [<xref ref-type="bibr" rid="ref23">23</xref>]–compatible structured representations, thus bridging the gap between unstructured text interpretation and structured, machine-readable outcome documentation.</p>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <sec>
        <title>Ethical Considerations</title>
        <p>This study was approved by the Ethics Committee of the Medical Faculty of the University of Duisburg-Essen (approval 23-11557-BO). Due to the study’s retrospective nature, the requirement of written informed consent was waived by the ethics committee. All data were fully anonymized before being included in the study.</p>
      </sec>
      <sec>
        <title>Dataset Collection</title>
        <p>Patients were eligible for inclusion if they had a liver resection between 2010 and 2024, their electronic health record contained at least 1 operative report confirming a hepatic surgical procedure, and at least 1 discharge letter documenting the perioperative course. Across all hepatobiliary service lines—hepatocellular carcinoma (HCC), colorectal liver metastases (CRLM), perihilar cholangiocarcinoma (Klatskin tumors), and intrahepatic cholangiocarcinoma (ICC)—a total of 650 postoperative cases met these criteria and were included after preprocessing (<xref rid="figure1" ref-type="fig">Figure 1</xref>).</p>
        <fig id="figure1" position="float">
          <label>Figure 1</label>
          <caption>
            <p>Flow diagram illustrating the stepwise identification of the study cohort. The figure shows successive screening and exclusion steps, resulting in a reduction from 1077 postoperative cases extracted from the hospital information system to 650 eligible cases with expert-assigned Clavien-Dindo grades used for model evaluation (created with BioRender). CRLM: colorectal liver metastases; HCC: hepatocellular carcinoma; Klatskin: perihilar cholangiocarcinoma; ICC: intrahepatic cholangiocarcinoma.</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e92407_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>For each patient, all available discharge and transfer letters were initially retrieved. As part of the expert annotation workflow, the final surgical discharge letter was manually selected for every patient. This expert-selected physician letter was treated as the authoritative postoperative summary and served as the sole input document for model prompting. During data preprocessing, letterhead sections were removed to standardize the narrative input.</p>
      </sec>
      <sec>
        <title>Expert Annotation</title>
        <p>Expert labeling was performed by a senior attending surgeon with &#62;10 years of clinical experience. For each case, the annotator was presented with all available discharge, transfer, and investigation letters from the relevant encounter, from which the surgical discharge letter was selected. The Clavien-Dindo grade was assigned exclusively based on the content of the selected letter, without access to additional clinical information such as laboratory values, imaging reports, or intraoperative findings, to ensure strict alignment with the information available to the language models (<xref rid="figure2" ref-type="fig">Figure 2</xref>).</p>
        <fig id="figure2" position="float">
          <label>Figure 2</label>
          <caption>
            <p>Expert annotation workflow for Clavien-Dindo grading. Schematic overview of the annotation process. For each case, the annotating surgeon reviewed all available discharge letters and selected the clinically most relevant document. Based on this letter and a standardized Clavien-Dindo definition displayed alongside, a complication grade was assigned. Optional free-text comments could be added to document uncertainty or case-specific considerations (created with BioRender).</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e92407_fig2.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>The manual annotation process was guided by a designated annotation user interface, which presented concise definitions of all Clavien-Dindo grades. Annotators assigned a single, predefined grade for each case using this structured interface, ensuring standardized input and minimizing formatting variability. Free-text comments could be added to flag ambiguous or borderline cases, facilitating subsequent review and, when necessary, refinement of the consensus labels. The resulting expert labels were used as the reference ground truth for all subsequent analyses.</p>
      </sec>
      <sec>
        <title>Interrater Reliability Assessment</title>
        <p>To assess the reliability of the expert annotation and to establish a human benchmark for model performance evaluation, a stratified random subset of 65 (10% of the full dataset) cases was selected for independent second annotation. Cases were sampled to approximate a balanced grade distribution across all Clavien-Dindo grades. Grades I-V were each represented by 8 cases. An exception was made for grade IVb, which occurred only 6 times in the full dataset and was therefore included in its entirety. Grade 0 was represented by 11 cases, reflecting its proportionally higher prevalence in the overall cohort.</p>
        <p>The second annotation was performed by an independent senior surgeon with 9 years of experience in hepatobiliary surgery and established familiarity with the Clavien-Dindo classification system. Both annotators were blinded to each other’s ratings, and the second annotator had no access to the primary expert annotations during the review process. Interrater agreement was quantified using Cohen κ, which accounts for agreement beyond chance. The resulting κ value serves as a human reference benchmark, allowing model-to-expert agreement, also reported as Cohen κ throughout the evaluation, to be interpreted in the context of clinician-to-clinician variability inherent to Clavien-Dindo grading.</p>
      </sec>
      <sec>
        <title>Model Selection</title>
        <p>For this study, a diverse set of LLMs was selected, encompassing both open-weight and proprietary architectures to assess performance across different model classes and deployment constraints (<xref ref-type="table" rid="table1">Table 1</xref>). Throughout this manuscript, the term “open-weight” refers to models whose trained parameters are publicly accessible and can be deployed locally, but whose training data, code, or full technical specifications may not be fully disclosed, distinguishing them from strictly “open-source” models, which imply complete transparency across all components. For the open-weight category, we selected 4 current LLMs: Qwen3 (Alibaba Cloud), Llama 3.3 (Meta AI), GPT-OSS (OpenAI), Ministral 3 (Mistral AI). These models were chosen for their strong general-purpose reasoning capabilities, broad community support, and the ability to deploy them locally, ensuring full compliance with institutional data protection regulations. Local deployment also enabled deterministic control over inference settings and resource allocation.</p>
        <table-wrap position="float" id="table1">
          <label>Table 1</label>
          <caption>
            <p>Large language models evaluated in this study. Qwen3, Llama 3.3, GPT-OSS, and Ministral 3 were locally deployed, while GPT-5.1 and Gemini 3 Pro were accessed via commercial APIs.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="410"/>
            <col width="390"/>
            <col width="200"/>
            <thead>
              <tr valign="top">
                <td>Model name</td>
                <td>Hardware specifications</td>
                <td>Open-weight</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Qwen/Qwen3-235B-A22B-Instruct-2507-FP8</td>
                <td>4x NVIDIA H100 GPU<sup>a</sup></td>
                <td>Yes</td>
              </tr>
              <tr valign="top">
                <td>meta-llama/Llama-3.3-70B-Instruct</td>
                <td>1x NVIDIA A100 GPU</td>
                <td>Yes</td>
              </tr>
              <tr valign="top">
                <td>openai/gpt-oss-120B</td>
                <td>1x NVIDIA H100 GPU</td>
                <td>Yes</td>
              </tr>
              <tr valign="top">
                <td>mistralai/Ministral-3-8B-Instruct-2512</td>
                <td>1x NVIDIA RTX A6000 GPU</td>
                <td>Yes</td>
              </tr>
              <tr valign="top">
                <td>gpt-5.1-2025-11-13</td>
                <td>Commercial deployment</td>
                <td>No</td>
              </tr>
              <tr valign="top">
                <td>gemini-3-pro-preview</td>
                <td>Commercial deployment</td>
                <td>No</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table1fn1">
              <p><sup>a</sup>GPU: graphics processing unit.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <p>To contextualize performance, we additionally included the 2 state-of-the-art proprietary models: GPT-5.1 and Gemini 3 Pro. These models were accessed via secure API end points (provider: OpenAI API) and selected for their strong benchmark performance in clinical reasoning and information extraction tasks [<xref ref-type="bibr" rid="ref24">24</xref>-<xref ref-type="bibr" rid="ref27">27</xref>].</p>
      </sec>
      <sec>
        <title>Prompting Approach</title>
        <p>Model interaction was based on a structured prompting framework, which we refined iteratively in consultation with the surgical expert to optimize clarity, task decomposition, and alignment with the Clavien-Dindo classification criteria. All evaluations were conducted in a zero-shot setting, without task-specific fine-tuning or exposure to labeled examples, to assess the intrinsic out-of-the-box capability of each model to infer postoperative complication severity directly from unstructured clinical documentation. This approach was deliberately chosen to establish a reproducible baseline that reflects the models’’ general reasoning ability rather than optimized task-specific performance. The final prompt consisted of 4 sequential components: a concise task description, a definition of the Clavien-Dindo scale, instructions for determining the correct grade, including a required output schema, and the full physician letter for each case. This structure ensured that models received identical contextual and instructional information. Full prompt templates are provided in Supplement S1-S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
        <p>Prompting was conducted under 2 distinct scenarios (<xref rid="figure3" ref-type="fig">Figure 3</xref>). In the first scenario, models were asked to predict the fine-grained Clavien-Dindo grade (0-V). In the second scenario, we derived a binary classification scheme in which the grades were grouped as <italic>minor complications</italic> (grades 0-IIIa) and <italic>major complications</italic> (grades IIIb-V). For each model output, we extracted 3 predefined fields, including score or category, reason, and citation, to enable structured downstream analysis. The same prompt was applied to all models without model-specific modifications. Additional context length constraints varied across models, and the Llama-3.3-70B was limited to approximately 6000 tokens, requiring truncation of a few excessively long physician letters. In such cases, appendices without additional clinical content were removed after manual review to retain the document’s clinically relevant core. All further hyperparameters are listed in Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
        <fig id="figure3" position="float">
          <label>Figure 3</label>
          <caption>
            <p>Large language model–based assessment of postoperative complications from electronic health records. Overview of the 2 evaluation scenarios applied to postoperative discharge letters. The upper part of the figure illustrates fine-grained prediction of Clavien-Dindo grades, while the lower part depicts binary classification into minor (≤IIIa) and major (≥IIIb) complication categories based on the same electronic health record inputs (created with BioRender).</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e92407_fig3.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Evaluation of Model Performance</title>
        <p>Model performance was assessed by comparing the Clavien-Dindo grades predicted by each LLM with the expert-assigned ground truth labels. Evaluation metrics included overall accuracy, grade-specific accuracy, entity-specific accuracy, precision, recall, <italic>F</italic><sub>1</sub>-scores (per grade, weighted <italic>F</italic><sub>1</sub> and macro <italic>F</italic><sub>1</sub>), and several measures of prediction error relative to expert grading. Accuracy, defined as the model’s agreement with the expert annotation, was calculated both across all cases and stratified by Clavien-Dindo grade, allowing assessment of whether model performance varied with complication severity. To account for potential heterogeneity across clinical indications, accuracy was also computed separately for each disease entity (HCC, CRLM, Klatskin tumors, and ICC). For this simplified task, we calculated overall and entity-specific accuracies, along with their corresponding confusion matrices, using ground-truth categories derived directly from the expert’s assigned Clavien-Dindo grades (<xref rid="figure4" ref-type="fig">Figure 4</xref>).</p>
        <fig id="figure4" position="float">
          <label>Figure 4</label>
          <caption>
            <p>End-to-end evaluation workflow for Clavien-Dindo classification. Schematic overview of the complete study pipeline, including expert annotation of postoperative discharge letters, structured prompting of large language models, quantitative performance evaluation, and manual review of cases with large prediction deviations for error analysis (created with BioRender). LLM: large language model.</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e92407_fig4.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>To quantify the magnitude of model-expert-disagreement, we calculated the absolute deviation, defined as the number of grade steps by which a model’s prediction differed from the ground truth. We report the distribution of deviation magnitudes, including the proportion of predictions differing by 1, 2, or more grade levels. These analyses were conducted overall and within each Clavien-Dindo grade, thereby enabling identification of grades that were particularly prone to over- or underestimation. In a final review step, outlier cases were defined as predictions deviating by more than 2 Clavien-Dindo grades from the expert-assigned reference and were manually reviewed. These cases were analyzed qualitatively to identify recurrent sources of disagreement, such as ambiguous documentation, implicit references to complication management, and inconsistent phrasing in physician letters. In addition, a majority voting approach was applied, aggregating predictions across the evaluated models and selecting the final grade based on the most frequent prediction. This ensemble strategy was used to assess whether combining model outputs could improve overall accuracy and reduce extreme disagreement with the expert annotation, under the assumption that individual models exhibit partially complementary error patterns, as suggested by the preceding error and deviation analyses.</p>
      </sec>
      <sec>
        <title>Proprietary Subset Selection</title>
        <p>For evaluating proprietary LLMs, a deidentified subset of 50 cases was created. This subset was intentionally balanced across the Clavien-Dindo grades, with 24 cases representing minor (0-IIIa) and 26 cases representing major (IIIb-V) complications. All documents underwent systematic deidentification [<xref ref-type="bibr" rid="ref28">28</xref>], replacing patient names, dates, institutional references, and location information in alignment with institutional data protection guidelines. This subset was used exclusively for evaluating GPT-5 and Gemini 3 Pro, whereas open-weight models were evaluated on the full dataset of 650 cases.</p>
      </sec>
      <sec>
        <title>Statistical Analysis</title>
        <p>Continuous variables were summarized as medians with IQRs. Before correlation analysis, normality of continuous variables was assessed using the Shapiro-Wilk test. As variables did not meet the assumption of normality, continuous clinical variables (eg, length of stay [LOS]) were analyzed descriptively and explored for association with complication severity using nonparametric rank-based correlation (Spearman ρ), given the ordinal structure of the Clavien-Dindo scale. To investigate potential sources of disagreement between model and expert annotation, logistic regression analyses with classification correctness (correct vs incorrect prediction) as the binary outcome variable were conducted. Document length, measured both in characters (per 1000 characters) and words (per 100 words), was included as the independent variable. Separate models were fitted for each evaluated LLM. Regression coefficients were exponentiated to yield odds ratios (ORs) with corresponding 95% CIs, quantifying the change in the odds of correct classification associated with incremental increases in document length. Statistical significance of individual predictors in logistic regression was evaluated using the Wald <italic>z</italic> test, and the significance level was set to <italic>P</italic>&#60;.05 for all analyses. Interrater agreement between the 2 independent clinical annotators, as well as agreement between each model and the primary expert annotation, was quantified using Cohen κ, calculated overall and separately for each Clavien-Dindo grade. κ values were interpreted according to the commonly used benchmarks (&#60;0.20 slight, 0.21-0.40 fair, 0.41-0.60 moderate, 0.61-0.80 substantial, and &#62;0.80 almost perfect agreement). All analyses were conducted in Python 3.12.9 (Python Software Foundation) using the packages <italic>pandas</italic> (2.3.1), <italic>numpy</italic> (2.3.1), and <italic>scipy</italic> (1.16.3), following a prespecified analytical plan to ensure reproducibility.</p>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <sec>
        <title>Cohort Characteristics</title>
        <p>The final dataset comprised 650 postoperative cases from 649 patients (n=229, 35% female) with a median age of 67 (IQR 58-73) years, treated between 2010 and 2024, and included 819 discharge letters. The cases cover the 4 major hepatobiliary disease entities: HCC (n=156, 24%), CRLM (n=113, 17%), perihilar cholangiocarcinoma (n=211, 33%), ICC (n=162, 25%), and other (n=8, 1%). The resulting expert annotation demonstrated a broad spectrum of postoperative complication severities within the study cohort. Most patients were classified as having no or minor complications, with grade 0 accounting for 334 (51%) cases, followed by grade I (n=90, 14%), grade II (n=63, 10%), and grade IIIa (n=63, 10%). Major complications were less frequent, including grade IIIb (n=41, 6%), grade IVa (n=11, 2%), grade IVb (n=6, 1%), and grade V (n=42, 6%). Overall, minor complications (grades 0-IIIa) comprised 85% (n=550) of the cohort, whereas major complications (grades IIIb-V) accounted for 15% (n=100; <xref ref-type="table" rid="table2">Table 2</xref>). The cohort characteristics of the 50-case subset are provided in Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
        <table-wrap position="float" id="table2">
          <label>Table 2</label>
          <caption>
            <p>Cohort characteristics stratified by Clavien-Dindo grade.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="30"/>
            <col width="130"/>
            <col width="120"/>
            <col width="120"/>
            <col width="100"/>
            <col width="100"/>
            <col width="90"/>
            <col width="100"/>
            <col width="100"/>
            <col width="110"/>
            <thead>
              <tr valign="top">
                <td colspan="2">Characteristics</td>
                <td colspan="8">Clavien-Dindo Grade</td>
              </tr>
              <tr valign="top">
                <td colspan="2">
                  <break/>
                </td>
                <td>0</td>
                <td>I</td>
                <td>II</td>
                <td>IIIa</td>
                <td>IIIb</td>
                <td>IVa</td>
                <td>IVb</td>
                <td>V</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td colspan="2">Age (y), median (IQR)</td>
                <td>66 (56.25-72)</td>
                <td>68 (59-72)</td>
                <td>71 (63-76)</td>
                <td>66 (59-71)</td>
                <td>71 (59-75)</td>
                <td>65 (58.5-67.5)</td>
                <td>72 (67-75.5)</td>
                <td>72 (65.25-75.75)</td>
              </tr>
              <tr valign="top">
                <td colspan="2">Total cases, n (%)</td>
                <td>334 (51)</td>
                <td>90 (14)</td>
                <td>63 (10)</td>
                <td>63 (10)</td>
                <td>41 (6)</td>
                <td>11 (2)</td>
                <td>6 (1)</td>
                <td>42 (6)</td>
              </tr>
              <tr valign="top">
                <td colspan="2">Total female, n (%)</td>
                <td>116 (35)</td>
                <td>34 (38)</td>
                <td>26 (41)</td>
                <td>28 (44)</td>
                <td>12 (29)</td>
                <td>3 (27)</td>
                <td>0 (0)</td>
                <td>10 (24)</td>
              </tr>
              <tr valign="top">
                <td colspan="10">Total cases per entity, n (%)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>ICC<sup>a</sup></td>
                <td>90 (27)</td>
                <td>30 (33)</td>
                <td>11 (17)</td>
                <td>16 (25)</td>
                <td>7 (17)</td>
                <td>3 (27)</td>
                <td>1 (17)</td>
                <td>4 (9)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>HCC<sup>b</sup></td>
                <td>89 (27)</td>
                <td>24 (27)</td>
                <td>9 (14)</td>
                <td>9 (14)</td>
                <td>4 (10)</td>
                <td>1 (9)</td>
                <td>2 (33)</td>
                <td>18 (43)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Klatskin</td>
                <td>78 (23)</td>
                <td>25 (28)</td>
                <td>30 (48)</td>
                <td>31 (49)</td>
                <td>20 (49)</td>
                <td>6 (55)</td>
                <td>1 (17)</td>
                <td>20 (48)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>CRLM<sup>c</sup></td>
                <td>71 (21)</td>
                <td>11 (12)</td>
                <td>12 (19)</td>
                <td>6 (10)</td>
                <td>10 (24)</td>
                <td>1 (9)</td>
                <td>2 (33)</td>
                <td>0 (0)</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Others</td>
                <td>6 (2)</td>
                <td>0 (0)</td>
                <td>1 (2)</td>
                <td>1 (2)</td>
                <td>0 (0)</td>
                <td>0 (0)</td>
                <td>0 (0)</td>
                <td>0 (0)</td>
              </tr>
              <tr valign="top">
                <td colspan="2">Year span</td>
                <td>2010-2024</td>
                <td>2010-2024</td>
                <td>2010-2024</td>
                <td>2011-2024</td>
                <td>2012-2024</td>
                <td>2013-2024</td>
                <td>2017-2023</td>
                <td>2011-2024</td>
              </tr>
              <tr valign="top">
                <td colspan="2">Text length in characters, median (IQR)</td>
                <td>4194 (3567.75-4814.5)</td>
                <td>4533.5 (3854-6175.25)</td>
                <td>5239 (4243-6305)</td>
                <td>5875 (4552.5-6995)</td>
                <td>6044 (4438-6971)</td>
                <td>7925 (5892-9316.5)</td>
                <td>8486 (7162.5-9709)</td>
                <td>4805 (4180.75-6934.25)</td>
              </tr>
              <tr valign="top">
                <td colspan="2">Text length in words, median (IQR)</td>
                <td>506 (419.5-581.75)</td>
                <td>553 (459-736.25)</td>
                <td>627 (497-776.5)</td>
                <td>705 (536.5-865)</td>
                <td>733 (504-842)</td>
                <td>1007 (696-1103.5)</td>
                <td>1050.5 (847-1215)</td>
                <td>555 (471-814.5)</td>
              </tr>
              <tr valign="top">
                <td colspan="2">LOS<sup>d</sup> (d), median (IQR)</td>
                <td>10 (8-14)</td>
                <td>15 (11-21)</td>
                <td>14 (11-23)</td>
                <td>23 (17-32.5)</td>
                <td>22 (19-32)</td>
                <td>42 (24-72)</td>
                <td>94.5 (42.25-100.25)</td>
                <td>11 (7.25-23)</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table2fn1">
              <p><sup>a</sup>ICC: intrahepatic cholangiocarcinoma.</p>
            </fn>
            <fn id="table2fn2">
              <p><sup>b</sup>HCC: hepatocellular carcinoma.</p>
            </fn>
            <fn id="table2fn3">
              <p><sup>c</sup>CRLM: colorectal liver metastases.</p>
            </fn>
            <fn id="table2fn4">
              <p><sup>d</sup>LOS: length of stay.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <p>The physician discharge letters exhibited high variability in length, with a median of 4558 (IQR 3797-5849.75) characters and 547 (IQR 450-697.75) words. Document length showed a weak-to-moderate positive association with complication severity (Spearman ρ=0.39 for characters and ρ=0.37 for words; both <italic>P</italic>&#60;.001). The median postoperative LOS was 13 (IQR 9-20) days and increased with higher Clavien-Dindo grades, demonstrating a moderate positive correlation with complication severity (Spearman ρ=0.48, <italic>P</italic>&#60;.001). LOS was also moderately correlated with document length (Spearman ρ=0.35 for characters and ρ=0.32 for words; both <italic>P</italic>&#60;.001), indicating that longer, more complex postoperative courses were accompanied by more extensive physician documentation. Given the distinct nature of Clavien-Dindo grade V, additional analyses excluding these cases are provided in Supplement S5 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
      </sec>
      <sec>
        <title>Interrater Reliability</title>
        <p>To contextualize model performance relative to human expert variability, interrater agreement between the 2 independent clinician annotations was assessed on the 65-case stratified subset. Overall, Cohen κ indicated substantial agreement between the 2 annotators (κ=0.752) for fine-grading and almost perfect agreement (κ=0.843) for binary grading, providing a human benchmark against which model-to-expert agreement can be interpreted. Grade-level results are presented in <xref ref-type="table" rid="table3">Table 3</xref>.</p>
        <table-wrap position="float" id="table3">
          <label>Table 3</label>
          <caption>
            <p>Interrater agreement between 2 independent clinical annotators across Clavien-Dindo grades on a 10% subset (65 cases).</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="310"/>
            <col width="200"/>
            <col width="490"/>
            <thead>
              <tr valign="top">
                <td>Grade</td>
                <td>Cases, n</td>
                <td>Interrater agreement Cohen κ</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Overall</td>
                <td>65</td>
                <td>0.752</td>
              </tr>
              <tr valign="top">
                <td>0</td>
                <td>11</td>
                <td>1.0</td>
              </tr>
              <tr valign="top">
                <td>I</td>
                <td>8</td>
                <td>0.871</td>
              </tr>
              <tr valign="top">
                <td>II</td>
                <td>8</td>
                <td>0.681</td>
              </tr>
              <tr valign="top">
                <td>IIIa</td>
                <td>8</td>
                <td>0.719</td>
              </tr>
              <tr valign="top">
                <td>IIIb</td>
                <td>8</td>
                <td>0.662</td>
              </tr>
              <tr valign="top">
                <td>IVa</td>
                <td>8</td>
                <td>0.405</td>
              </tr>
              <tr valign="top">
                <td>IVb</td>
                <td>6</td>
                <td>0.476</td>
              </tr>
              <tr valign="top">
                <td>V</td>
                <td>8</td>
                <td>0.932</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <p>Agreement varied considerably across grades. Perfect agreement was observed for grade 0 (κ=1.000) and almost perfect agreement for grade V (κ=0.932) and grade I (κ=0.871), reflecting the relative clarity of these categories. Substantial agreement was found for grades II, IIIa, and IIIb (κ=0.681-0.719), while grades IVa and IVb showed the lowest interrater concordance (κ=0.405 and κ=0.476, respectively), consistent with the known ambiguity in clinical documentation. These findings are in line with previously reported interrater variability for Clavien-Dindo grading, where comparable levels of agreement for complex complication grades have been observed in urological surgery [<xref ref-type="bibr" rid="ref29">29</xref>].</p>
      </sec>
      <sec>
        <title>Performance of Open-Weight Models</title>
        <sec>
          <title>Results for Fine-Grained Prediction</title>
          <p>The open-weight models Qwen3-235B, Llama-3.3-70B, GPT-OSS 120B, and Ministral-3-8B were evaluated on the full dataset of 650 cases in a zero-shot setting. Overall accuracy in predicting the fine-grained Clavien-Dindo grade ranged from 0.749 to 0.775, with Qwen3-235B achieving the highest overall accuracy (0.775) among the open-weight models (<xref ref-type="table" rid="table4">Table 4</xref>). To provide a more comprehensive picture of model performance on the class-imbalanced dataset, macroaveraged and weighted <italic>F</italic><sub>1</sub>-scores were additionally calculated. Macroaveraged <italic>F</italic><sub>1</sub>-scores ranged from 0.499 (Ministral-3-8B) to 0.627 (Qwen3-235B), reflecting the greater challenge of correctly classifying underrepresented grades. Weighted <italic>F</italic><sub>1</sub>-scores, which account for class frequency, ranged from 0.758 to 0.779 and were more closely aligned with overall accuracy. Per-grade precision and recall values are provided in Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
          <table-wrap position="float" id="table4">
            <label>Table 4</label>
            <caption>
              <p>Grade-level and overall classification performance for fine-grained Clavien-Dindo grade prediction across 4 open-weight large language models evaluated on the full 650-case dataset in a zero-shot setting. For each Clavien-Dindo grade, the number of cases, accuracy, and F1-score are reported per model. Overall performance is summarized as overall accuracy, macroaveraged F1-score (unweighted mean across all grades, reflecting performance on rare and common grades equally), and weighted averaged F1-score (weighted by class frequency, accounting for the imbalanced grade distribution).</p>
            </caption>
            <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
              <col width="80"/>
              <col width="80"/>
              <col width="110"/>
              <col width="100"/>
              <col width="0"/>
              <col width="110"/>
              <col width="100"/>
              <col width="0"/>
              <col width="110"/>
              <col width="100"/>
              <col width="0"/>
              <col width="110"/>
              <col width="100"/>
              <thead>
                <tr valign="bottom">
                  <td>Grade</td>
                  <td>Cases, n</td>
                  <td colspan="3">Qwen3-235B<sup>a</sup></td>
                  <td colspan="3">Llama-3.3-70B<sup>b</sup></td>
                  <td colspan="3">GPT-OSS 120B<sup>c</sup></td>
                  <td colspan="2">Ministral 3-8B<sup>d</sup></td>
                </tr>
                <tr valign="top">
                  <td>
                    <break/>
                  </td>
                  <td>
                    <break/>
                  </td>
                  <td>Accuracy</td>
                  <td><italic>F</italic><sub>1</sub>-score</td>
                  <td colspan="2">Accuracy</td>
                  <td><italic>F</italic><sub>1</sub>-score</td>
                  <td colspan="2">Accuracy</td>
                  <td><italic>F</italic><sub>1</sub>-score</td>
                  <td colspan="2">Accuracy</td>
                  <td><italic>F</italic><sub>1</sub>-score</td>
                </tr>
              </thead>
              <tbody>
                <tr valign="top">
                  <td>0</td>
                  <td>334</td>
                  <td>0.904</td>
                  <td>0.921</td>
                  <td colspan="2">0.853</td>
                  <td>0.896</td>
                  <td colspan="2">0.880</td>
                  <td>0.899</td>
                  <td colspan="2">0.961</td>
                  <td>0.920</td>
                </tr>
                <tr valign="top">
                  <td>I</td>
                  <td>90</td>
                  <td>0.5</td>
                  <td>0.559</td>
                  <td colspan="2">0.556</td>
                  <td>0.553</td>
                  <td colspan="2">0.422</td>
                  <td>0.507</td>
                  <td colspan="2">0.322</td>
                  <td>0.453</td>
                </tr>
                <tr valign="top">
                  <td>II</td>
                  <td>63</td>
                  <td>0.667</td>
                  <td>0.56</td>
                  <td colspan="2">0.746</td>
                  <td>0.531</td>
                  <td colspan="2">0.714</td>
                  <td>0.608</td>
                  <td colspan="2">0.629</td>
                  <td>0.538</td>
                </tr>
                <tr valign="top">
                  <td>IIIa</td>
                  <td>63</td>
                  <td>0.571</td>
                  <td>0.567</td>
                  <td colspan="2">0.429</td>
                  <td>0.551</td>
                  <td colspan="2">0.810</td>
                  <td>0.694</td>
                  <td colspan="2">0.587</td>
                  <td>0.607</td>
                </tr>
                <tr valign="top">
                  <td>IIIb</td>
                  <td>41</td>
                  <td>0.780</td>
                  <td>0.753</td>
                  <td colspan="2">0.707</td>
                  <td>0.707</td>
                  <td colspan="2">0.756</td>
                  <td>0.775</td>
                  <td colspan="2">0.659</td>
                  <td>0.72</td>
                </tr>
                <tr valign="top">
                  <td>IVa</td>
                  <td>11</td>
                  <td>0.273</td>
                  <td>0.222</td>
                  <td colspan="2">0.636</td>
                  <td>0.412</td>
                  <td colspan="2">0.182</td>
                  <td>0.133</td>
                  <td colspan="2">0.636</td>
                  <td>0.341</td>
                </tr>
                <tr valign="top">
                  <td>IVb</td>
                  <td>6</td>
                  <td>0.333</td>
                  <td>0.444</td>
                  <td colspan="2">0.167</td>
                  <td>0.286</td>
                  <td colspan="2">0.0</td>
                  <td>0.0</td>
                  <td colspan="2">0.333</td>
                  <td>0.444</td>
                </tr>
                <tr valign="top">
                  <td>V</td>
                  <td>42</td>
                  <td>1.0</td>
                  <td>0.988</td>
                  <td colspan="2">0.976</td>
                  <td>0.965</td>
                  <td colspan="2">0.976</td>
                  <td>0.976</td>
                  <td colspan="2">0.952</td>
                  <td>0.964</td>
                </tr>
              </tbody>
            </table>
            <table-wrap-foot>
              <fn id="table4fn1">
                <p><sup>a</sup>Overall accuracy=0.775, macroaveraged <italic>F</italic><sub>1</sub>=0.627, and weighted averaged <italic>F</italic><sub>1</sub>=0.779.</p>
              </fn>
              <fn id="table4fn2">
                <p><sup>b</sup>Overall accuracy=0.749, macroaveraged <italic>F</italic><sub>1</sub>=0.613, and weighted averaged <italic>F</italic><sub>1</sub>=0.758.</p>
              </fn>
              <fn id="table4fn3">
                <p><sup>c</sup>Overall accuracy=0.772, macroaveraged <italic>F</italic><sub>1</sub>=0.574, and weighted averaged <italic>F</italic><sub>1</sub>=0.773.</p>
              </fn>
              <fn id="table4fn4">
                <p><sup>d</sup>Overall accuracy=0.768, macroaveraged <italic>F</italic><sub>1</sub>=0.499, and weighted averaged <italic>F</italic><sub>1</sub>=0.764.</p>
              </fn>
            </table-wrap-foot>
          </table-wrap>
          <p>Grade-specific analysis revealed differences across severity levels (<xref ref-type="table" rid="table4">Table 4</xref>). Performance was highest for grade 0 (accuracy=0.853-0.961; <italic>F</italic><sub>1</sub>-score=0.896-0.921) and grade V (accuracy=0.952-1.000; <italic>F</italic><sub>1</sub>-score=0.964-0.988), whereas predictive performance declined for intermediate and less frequent grades, such as grade I (accuracy=0.322-0.556; <italic>F</italic><sub>1</sub>-score=0.453-0.559), grade IVa (accuracy=0.182-0.636; <italic>F</italic><sub>1</sub>-score=0.133-0.412), and grade IVb (accuracy=0.000-0.333; <italic>F</italic><sub>1</sub>-score=0.000-0.444).</p>
          <p>Model-specific performance differences were observed across Clavien-Dindo grades. Qwen3-235B demonstrated the most balanced performance, achieving particularly high accuracy for grades 0, IIIb, and V. Llama-3.3-70B performed well for common complications, but showed reduced accuracy for rare, severe events. GPT-OSS-120B achieved high accuracy for grade IIIa (0.810) and demonstrated consistent performance across most grades. By contrast, Ministral-3-8B exhibited greater instability, producing “not classifiable” outputs in 5 cases (omission rate=0.8%), with abstentions distributed across grades 0, II, and IIIa. The impact on overall accuracy was minimal; excluding unclassified cases from the denominator yielded an accuracy of 0.774, while treating them as prediction errors resulted in an adjusted accuracy of 0.768 (Table S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Entity-specific accuracy (Table S5 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) showed moderate variation across clinical subgroups. Performance was highest in CRLM cases (accuracy range=0.814-0.832), whereas Klatskin cases exhibited the lowest accuracy (0.687-0.768). Accuracy for HCC and ICC cases was comparable across models, with values between 0.735 and 0.827.</p>
          <p>To assess potential temporal trends across the 14-year observation period, model performance was additionally stratified by year of discharge letter and is provided in Table S6 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. A stratified analysis comparing minor complications (≤grade IIIa) and major complications (≥grade IIIb) demonstrated similar performance across models. The accuracy for minor complications ranged from 0.74 to 0.78, while major complications showed accuracy values between 0.74 and 0.79, indicating that open-weight models maintained relatively stable performance across the complication severity spectrum. Corresponding confusion matrices for the binary grouping (≤IIIa vs ≥IIIb) are provided in Figure S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
          <p>Cohen κ was calculated between each model’s predictions and the primary expert annotation on the 65-case subset, enabling direct comparison with the previously established human interrater benchmark (κ=0.752). Results are presented in <xref ref-type="table" rid="table5">Table 5</xref>.</p>
          <table-wrap position="float" id="table5">
            <label>Table 5</label>
            <caption>
              <p>Model-to-expert agreement quantified as Cohen κ for fine-grained Clavien-Dindo grade prediction across 4 open-weight large language models, evaluated on the stratified 65-case subset. Cohen κ is reported overall and separately for each Clavien-Dindo grade, allowing direct comparison with the human interrater benchmark (κ=0.752) established on the same subset. Negative κ values indicate agreement below chance level. κ values were interpreted as follows: &#60;0.20 slight, 0.21-0.40 fair, 0.41-0.60 moderate, 0.61-0.80 substantial, and &#62;0.80 almost perfect agreement.</p>
            </caption>
            <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
              <col width="140"/>
              <col width="100"/>
              <col width="190"/>
              <col width="190"/>
              <col width="190"/>
              <col width="190"/>
              <thead>
                <tr valign="top">
                  <td>Grade</td>
                  <td>Cases, n</td>
                  <td>Qwen3-235B</td>
                  <td>Llama-3.3-70B</td>
                  <td>GPT-OSS 120B</td>
                  <td>Ministral 3-8B</td>
                </tr>
              </thead>
              <tbody>
                <tr valign="top">
                  <td>Overall</td>
                  <td>65</td>
                  <td>0.656</td>
                  <td>0.588</td>
                  <td>0.541</td>
                  <td>0.663</td>
                </tr>
                <tr valign="top">
                  <td>0</td>
                  <td>11</td>
                  <td>0.891</td>
                  <td>0.816</td>
                  <td>0.781</td>
                  <td>0.947</td>
                </tr>
                <tr valign="top">
                  <td>I</td>
                  <td>8</td>
                  <td>0.925</td>
                  <td>0.797</td>
                  <td>0.774</td>
                  <td>0.745</td>
                </tr>
                <tr valign="top">
                  <td>II</td>
                  <td>8</td>
                  <td>0.662</td>
                  <td>0.502</td>
                  <td>0.662</td>
                  <td>0.662</td>
                </tr>
                <tr valign="top">
                  <td>IIIa</td>
                  <td>8</td>
                  <td>0.485</td>
                  <td>0.405</td>
                  <td>0.693</td>
                  <td>0.485</td>
                </tr>
                <tr valign="top">
                  <td>IIIb</td>
                  <td>8</td>
                  <td>0.743</td>
                  <td>0.572</td>
                  <td>0.570</td>
                  <td>0.572</td>
                </tr>
                <tr valign="top">
                  <td>IVa</td>
                  <td>8</td>
                  <td>0.201</td>
                  <td>0.485</td>
                  <td>0.021</td>
                  <td>0.485</td>
                </tr>
                <tr valign="top">
                  <td>IVb</td>
                  <td>6</td>
                  <td>0.408</td>
                  <td>0.266</td>
                  <td>–0.027</td>
                  <td>0.486</td>
                </tr>
                <tr valign="top">
                  <td>V</td>
                  <td>8</td>
                  <td>0.932</td>
                  <td>0.857</td>
                  <td>0.857</td>
                  <td>0.932</td>
                </tr>
              </tbody>
            </table>
          </table-wrap>
          <p>Overall model-to-expert agreement was moderate to substantial for all evaluated open-weight models, with κ values ranging from 0.541 (GPT-OSS) to 0.663 (Ministral 3-8B). All models fell below the overall human interrater benchmark.</p>
          <p>Grade-level analysis revealed consistent patterns across models. Agreement was highest for grade 0 (κ=0.781-0.947), grade I (κ=0.745-0.925), and grade V (κ=0.857-0.932), mirroring the pattern observed in human interrater agreement, where these grades were also most consistently classified. Grades II, IIIa, and IIIb showed moderate to substantial agreement across all models (κ=0.405-0.743), comparable with the human interrater agreement for these grades (κ=0.681, κ=0.719, and κ=0.662, respectively). The most challenging grades were IVa and IVb, where agreement was low across all models (κ=0.021-0.485 and κ=−0.027 to 0.486, respectively), with GPT-OSS 120B showing near-zero agreement, indicating performance at chance level for this category, consistent with the relatively low human interrater agreement for these grades as well (κ=0.405 and κ=0.476). Notably, several models approached or exceeded the human interrater benchmark.</p>
        </sec>
        <sec>
          <title>Results for Binary Grading</title>
          <p>In the binary grading task, all open-weight models achieved higher accuracy than in fine-grade prediction. Overall model accuracy ranged from 0.928 (Ministral-3-8B) to 0.94 (Llama-3.3-70B). Although accuracy metrics were closely clustered across models, the <italic>F</italic><sub>1</sub>-scores provided additional insight into classification robustness. All models achieved <italic>F</italic><sub>1</sub>-scores between 0.96 and 0.97, indicating a very strong balance between precision and recall in the binary setting. Confusion matrices for the binary classification task demonstrated that disagreement with expert annotation occurred predominantly as category A predicted instead of B, rather than the reverse, reflecting conservative tendencies in model outputs (Figure S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p>
          <p>Omission rates varied substantially across models in the binary classification task (Table S7 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). While Qwen3-235B, Llama-3.3-70B, and GPT-OSS-120B produced very few or no unclassified outputs (omission rates=0.5%, 0%, and 0.3%, respectively), Ministral-3-8B declined to classify 36 cases (omission rate=5.5%), all falling into category A. When unclassified cases were treated as prediction errors, Ministral-3-8B’s adjusted accuracy dropped from 0.928 to 0.877, representing the largest impact of omission rate on overall performance across all models.</p>
          <p>To evaluate whether model performance varied across clinical indications, accuracy was calculated separately for each hepatobiliary disease entity (Table S8 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The observed performance was highest in CRLM cases, with accuracies ranging from 0.965 to 0.991. Similarly, strong performance was observed for HCC (0.949-0.962) and ICC (0.928-0.951) across all models. The “Other” category showed perfect accuracy in 3 of 4 models, though its small sample size limits generalizability. Performance in Klatskin tumors was comparatively lower, with accuracies between 0.883 and 0.905. In addition to quantitative accuracy metrics, we examined how often models were unable to assign any Clavien-Dindo grade. Such “not classifiable” outputs were rare in the larger open-weight models but were markedly more frequent in the smallest model, Ministral-3-8B, which produced 36 unclassified cases. Because these outputs do not correspond to valid Clavien-Dindo categories, they were excluded from grade-level accuracy calculations but included in qualitative error assessment. To assess the potential influence of temporal trends on model performance, accuracy and weighted <italic>F</italic><sub>1</sub>-scores were additionally calculated per year across the full 2010-2024 observation period (Table S9 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p>
          <p>Extended evaluation metrics for the binary grading task are provided in Table S10 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Per-class analysis revealed a consistent asymmetry across all models: category A (no or minor complication) was classified with high precision (0.979-0.985) but comparatively lower recall, while category B (major complication) showed the inverse pattern, with higher recall (0.889-0.920) but substantially lower precision (0.719-0.772). This reflects a tendency of all models to over-predict grade B, resulting in fewer missed major complications at the cost of more false positives. Macroaveraged <italic>F</italic><sub>1</sub>-scores ranged from 0.882 (Ministral 3-8B) to 0.897 (Qwen3-235B), and weighted <italic>F</italic><sub>1</sub>-scores from 0.932 to 0.945, confirming robust overall performance while accounting for class imbalance. Model-to-expert agreement, quantified as Cohen κ, ranged from κ=0.815 (Ministral 3-8B) to κ=0.877 (Qwen3-235B), indicating almost perfect agreement and exceeding the human interrater benchmark of κ=0.843 established for binary annotation.</p>
        </sec>
      </sec>
      <sec>
        <title>Performance of Proprietary Models</title>
        <p>To evaluate the performance of proprietary LLMs under similar conditions, GPT-5.1 and Gemini 3 Pro Preview were prompted on a balanced, deidentified subset of 50 cases selected from the full dataset, ensuring uniform representation across all complication severities. While the 4 open-weight models were primarily evaluated on the full 650-case dataset as reported above, their performance was additionally assessed on the same 50-case subset to provide a methodologically equivalent basis for cross-model comparison. The full-dataset results for open-weight models and subset results for proprietary and open-weight models are not directly comparable. The subset-based results for both proprietary and open-weight models are presented side by side in Tables S11-S15 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>, including grade-level accuracy, <italic>F</italic><sub>1</sub>-scores, entity-specific accuracy, deviation distributions, and confusion matrices.</p>
        <p>GPT-5.1 and Gemini 3 Pro each achieved an overall accuracy of 0.78, with macroaveraged <italic>F</italic><sub>1</sub>-scores of 0.766 and 0.751, and weighted <italic>F</italic><sub>1</sub>-scores of 0.774 and 0.761, respectively (<xref ref-type="table" rid="table6">Table 6</xref>). By comparison, open-weight models achieved overall accuracies of 0.60-0.70 on the same subset, with macroaveraged <italic>F</italic><sub>1</sub>-scores ranging from 0.557 to 0.673 and weighted <italic>F</italic><sub>1</sub>-scores from 0.570 to 0.688 (Table S11 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p>
        <table-wrap position="float" id="table6">
          <label>Table 6</label>
          <caption>
            <p>Grade-level and overall classification accuracy and F1-scores for fine-grained Clavien-Dindo grade prediction by 2 proprietary large language models (GPT-5.1 and Gemini 3 Pro Preview), evaluated on a balanced, deidentified 50-case subset in a zero-shot setting. Direct comparison with open-weight model performance on the same subset is provided in Table S11 in Multimedia Appendix 1.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="160"/>
            <col width="170"/>
            <col width="160"/>
            <col width="200"/>
            <col width="0"/>
            <col width="160"/>
            <col width="150"/>
            <thead>
              <tr valign="bottom">
                <td>Grade</td>
                <td>Cases, n</td>
                <td colspan="3">GPT 5.1<sup>a</sup></td>
                <td colspan="2">Gemini 3 Pro preview<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>
                  <break/>
                </td>
                <td>Accuracy</td>
                <td><italic>F</italic><sub>1</sub>-score</td>
                <td colspan="2">Accuracy</td>
                <td><italic>F</italic><sub>1</sub>-score</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>0</td>
                <td>6</td>
                <td>0.833</td>
                <td>0.833</td>
                <td colspan="2">1.0</td>
                <td>0.923</td>
              </tr>
              <tr valign="top">
                <td>I</td>
                <td>7</td>
                <td>1.0</td>
                <td>0.933</td>
                <td colspan="2">1.0</td>
                <td>0.875</td>
              </tr>
              <tr valign="top">
                <td>II</td>
                <td>5</td>
                <td>0.6</td>
                <td>0.667</td>
                <td colspan="2">0.6</td>
                <td>0.6</td>
              </tr>
              <tr valign="top">
                <td>IIIa</td>
                <td>6</td>
                <td>0.833</td>
                <td>0.714</td>
                <td colspan="2">0.833</td>
                <td>0.833</td>
              </tr>
              <tr valign="top">
                <td>IIIb</td>
                <td>7</td>
                <td>0.714</td>
                <td>0.714</td>
                <td colspan="2">0.857</td>
                <td>0.800</td>
              </tr>
              <tr valign="top">
                <td>IVa</td>
                <td>6</td>
                <td>0.667</td>
                <td>0.667</td>
                <td colspan="2">0.333</td>
                <td>0.444</td>
              </tr>
              <tr valign="top">
                <td>IVb</td>
                <td>6</td>
                <td>0.5</td>
                <td>0.667</td>
                <td colspan="2">0.5</td>
                <td>0.6</td>
              </tr>
              <tr valign="top">
                <td>V</td>
                <td>7</td>
                <td>1.0</td>
                <td>0.933</td>
                <td colspan="2">1.0</td>
                <td>0.933</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table6fn1">
              <p><sup>a</sup>Overall accuracy=0.78, macroaveraged <italic>F</italic><sub>1</sub>=0.766, and weighted averaged <italic>F</italic><sub>1</sub>=0.774.</p>
            </fn>
            <fn id="table6fn2">
              <p><sup>b</sup>Overall accuracy=0.78, macroaveraged <italic>F</italic><sub>1</sub>=0.751, and weighted averaged <italic>F</italic><sub>1</sub>=0.761.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <p>Moreover, both proprietary models achieved perfect accuracy of 1.0 for grades I and V. The performance for grades IIIa and IIIb was also strong, with values ranging between 0.714 and 0.857 and <italic>F</italic><sub>1</sub>-scores from 0.714 to 0.833. In contrast, grade II showed mixed performance, with an accuracy of 0.6 and <italic>F</italic><sub>1</sub>-scores of 0.600-0.667, similar to the results of the open-weight models. For the rarer grades IVa and IVb, greater variability was observed, with accuracies ranging between 0.333 and 0.667 and <italic>F</italic><sub>1</sub>-scores between 0.444 and 0.667.</p>
        <p>Overall, GPT-5.1 and Gemini 3 Pro demonstrated performance that was partly superior, partly comparable, and in some cases lower than the performance of the strongest open-weight models, depending on the complication grade and clinical entity. Both proprietary models showed clear strengths in several grades, including robust performance in grades I, IIIa, IIIb, and V, while exhibiting similar or slightly reduced performance in other grades, particularly some intermediate or low-frequency grades. When complications were grouped into minor (≤IIIa) and major (≥IIIb) categories, both proprietary models outperformed all open-weight models, achieving higher accuracy in distinguishing clinically relevant severity classes (Table S12 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Entity-level performance was consistently high for both proprietary models. Compared with open-weight baselines, proprietary models performed better or at least as well across all entities; particularly strong gains were observed in Klatskin cases, which involve complex postoperative trajectories and narrative heterogeneity (Table S13 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). These results indicate that proprietary models generalize more robustly across diverse disease contexts.</p>
        <p>In the binary classification task, proprietary models again achieved the highest performance. On the 50-case subset, GPT-5.1 reached an accuracy of 0.98 and a macroaveraged <italic>F</italic><sub>1</sub>-score of 0.980, with perfect recall for category A (1.00) and perfect precision for category B (1.00). Gemini 3 Pro Preview achieved an accuracy of 0.94 and a macroaveraged <italic>F</italic><sub>1</sub>-score of 0.940, with a recall of 0.958 for category A and 0.923 for category B. By comparison, open-weight models achieved accuracies of 0.90-0.94 on the same subset, with macroaveraged <italic>F</italic><sub>1</sub>-scores of 0.900-0.940 (Table S14 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Entity-level binary classification accuracy for GPT-5.1 and Gemini was uniformly high, ranging from 0.8 to 1.0 across all clinical subgroups, confirming that distinguishing minor from major complications is a highly tractable task for state-of-the-art LLMs (Table S15 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p>
      </sec>
      <sec>
        <title>Analysis of Classification Errors</title>
        <sec>
          <title>Predictors of Classification Accuracy</title>
          <p>To explore potential sources of disagreement with expert annotation, we examined whether characteristics of the input text were associated with prediction errors. Specifically, we fitted a logistic regression model with classification correctness (correct vs incorrect) as the dependent variable and document length measured in both characters and words as independent variables. A detailed summary of the regression coefficients is provided in <xref ref-type="table" rid="table7">Table 7</xref>.</p>
          <table-wrap position="float" id="table7">
            <label>Table 7</label>
            <caption>
              <p>Association between discharge letter length and classification accuracy across all evaluated large language models, reported separately for open-weight models (n=650; Ministral-3-8B n=645 due to unclassified outputs) and proprietary models (n=50, balanced subset). Logistic regression results are presented as odds ratios with 95% CIs and <italic>P</italic> values, calculated per 1000 characters and per 100 words.</p>
            </caption>
            <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
              <col width="130"/>
              <col width="70"/>
              <col width="330"/>
              <col width="100"/>
              <col width="270"/>
              <col width="100"/>
              <thead>
                <tr valign="bottom">
                  <td>Model</td>
                  <td>Cases, n</td>
                  <td>OR<sup>a</sup> (95% CI) per 1000 characters</td>
                  <td><italic>P</italic> value</td>
                  <td>OR (95% CI) per 100 words</td>
                  <td><italic>P</italic> value</td>
                </tr>
              </thead>
              <tbody>
                <tr valign="top">
                  <td>Qwen3</td>
                  <td>650</td>
                  <td>0.83 (0.75-0.91)</td>
                  <td>&#60;.001</td>
                  <td>0.86 (0.79-0.93)</td>
                  <td>&#60;.001</td>
                </tr>
                <tr valign="top">
                  <td>Llama 3.3</td>
                  <td>650</td>
                  <td>0.90 (0.82-0.99)</td>
                  <td>.02</td>
                  <td>0.91 (0.85-0.98)</td>
                  <td>.02</td>
                </tr>
                <tr valign="top">
                  <td>GPT-OSS</td>
                  <td>650</td>
                  <td>0.87 (0.79-0.96)</td>
                  <td>.005</td>
                  <td>0.90 (0.83-0.97)</td>
                  <td>.008</td>
                </tr>
                <tr valign="top">
                  <td>Ministral 3</td>
                  <td>645</td>
                  <td>0.86 (0.78-0.95)</td>
                  <td>.002</td>
                  <td>0.89 (0.82-0.96)</td>
                  <td>.003</td>
                </tr>
                <tr valign="top">
                  <td>GPT 5.1</td>
                  <td>50</td>
                  <td>0.93 (0.71-1.2)</td>
                  <td>.57</td>
                  <td>0.96 (0.78-1.17)</td>
                  <td>.68</td>
                </tr>
                <tr valign="top">
                  <td>Gemini 3</td>
                  <td>50</td>
                  <td>0.73 (0.54-0.99)</td>
                  <td>.04</td>
                  <td>0.80 (0.63-1.00)</td>
                  <td>.049</td>
                </tr>
              </tbody>
            </table>
            <table-wrap-foot>
              <fn id="table7fn1">
                <p><sup>a</sup>OR: odds ratio.</p>
              </fn>
            </table-wrap-foot>
          </table-wrap>
          <p>The OR quantifies the change in odds of correct classification per unit increase in the predictor variable (1000 characters or 100 words). For the Qwen3 model, an OR of 0.83 indicates that for each additional 1000 characters, the odds of a correct classification decrease by 17%.</p>
          <p>Across most evaluated models, increasing document length was associated with lower odds of correct classification. OR below 1.0 indicates that, per incremental increase in text length, the likelihood of a correct Clavien-Dindo assignment decreased. This effect was statistically significant for all open-weight models and for Gemini 3 Pro, whereas no significant association was observed for GPT-5.1, likely due to the limited number of samples in the proprietary-model subset. These findings suggest that longer discharge letters may introduce additional narrative complexity or ambiguity that complicates model interpretation, rather than uniformly providing more discriminative information.</p>
        </sec>
        <sec>
          <title>Deviation Analysis</title>
          <p>To quantify the magnitude of disagreement between model predictions and expert annotations, we computed the absolute deviation defined as the number of grade levels separating the predicted grade from the reference label (<xref ref-type="table" rid="table8">Table 8</xref>).</p>
          <table-wrap position="float" id="table8">
            <label>Table 8</label>
            <caption>
              <p>Distribution of absolute grade deviations between model predictions and primary expert annotation for 4 open-weight large language models, evaluated on the full dataset. Absolute deviation is defined as the number of Clavien-Dindo grade levels separating the predicted grade from the reference label, regardless of direction. Only cases with incorrect predictions are included. Values are reported as absolute case counts (n) and relative proportions (%) of all incorrectly classified cases per model.</p>
            </caption>
            <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
              <col width="160"/>
              <col width="210"/>
              <col width="210"/>
              <col width="210"/>
              <col width="210"/>
              <thead>
                <tr valign="top">
                  <td>Absolute deviation</td>
                  <td>Qwen3-235B, n (%)</td>
                  <td>Llama-3.3-70B, n (%)</td>
                  <td>GPT-OSS 120B, n (%)</td>
                  <td>Ministral-3-8B, n (%)</td>
                </tr>
              </thead>
              <tbody>
                <tr valign="top">
                  <td>1</td>
                  <td>96 (66.21)</td>
                  <td>117 (71.78)</td>
                  <td>83 (56.08)</td>
                  <td>96 (65.75)</td>
                </tr>
                <tr valign="top">
                  <td>2</td>
                  <td>37 (25.52)</td>
                  <td>33 (20.25)</td>
                  <td>42 (28.38)</td>
                  <td>27 (18.49)</td>
                </tr>
                <tr valign="top">
                  <td>3</td>
                  <td>7 (4.83)</td>
                  <td>4 (2.45)</td>
                  <td>15 (10.14)</td>
                  <td>12 (8.22)</td>
                </tr>
                <tr valign="top">
                  <td>4</td>
                  <td>5 (3.45)</td>
                  <td>8 (4.91)</td>
                  <td>5 (3.38)</td>
                  <td>8 (5.48)</td>
                </tr>
                <tr valign="top">
                  <td>5</td>
                  <td>—<sup>a</sup></td>
                  <td>1 (0.61)</td>
                  <td>3 (2.03)</td>
                  <td>2 (1.37)</td>
                </tr>
                <tr valign="top">
                  <td>8</td>
                  <td>—</td>
                  <td>—</td>
                  <td>—</td>
                  <td>1 (0.68)</td>
                </tr>
              </tbody>
            </table>
            <table-wrap-foot>
              <fn id="table8fn1">
                <p><sup>a</sup>Not available.</p>
              </fn>
            </table-wrap-foot>
          </table-wrap>
          <p>Across open-weight models, 1-step deviations accounted for 55.29% to 71.17%, representing the majority of disagreement with expert annotation. Deviations with 3 or more steps were comparatively rare, occurring in 7.97%-15,75% of prediction errors across models. Grade-specific deviation distributions indicated that grade 0 predictions exhibited the fewest errors, whereas grades II-IVb demonstrated higher variability in error magnitude. Detailed deviation profiles per open-weight model and per grade are shown in Table S16 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
          <p>Additionally, the deviation analysis was conducted on the subset with open-weight and proprietary models, revealing systematic differences. While GPT-5.1 and Gemini 3 Pro produced fewer 1-step deviations than most open-weight systems, they showed a higher proportion of 3-step deviations, indicating that although their predictions were often correct, their errors occasionally involved larger jumps in severity. For 2-step deviations, proprietary and open-weight models were broadly similar. However, GPT-OSS 120B represented an exception to this pattern; its deviation distribution was equal to or worse than that of the proprietary models across several grades, with relatively few 1-step deviations and a comparatively high proportion of larger errors. This trend was further confirmed by the per-grade deviation analysis, where GPT-OSS 120B displayed a deviation distribution more similar to that of the proprietary models than to the stronger open-weight systems. This pattern reflects a more polarized deviation distribution in the proprietary models; they were either correct or close in many cases, but when incorrect, they were more likely than open-weight models to misclassify by a larger margin. Full deviation distributions are shown in Tables S17 and S18 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
        </sec>
        <sec>
          <title>Analysis of Reasons for High Deviation</title>
          <p>Cases with high deviation rates (&#62;2 categories) underwent separate manual analysis. Among these, in 10 cases, prolonged intensive care unit (ICU) monitoring without any further therapy was detected as a life-threatening complication (Clavien-Dindo grade≥IVa), and in 7 cases, postoperative intervention was classified as part of the regular oncologic therapy and not as a surgical complication. Renal failure and liver failure without any specific therapy were also among the reasons for high deviation grades. For further deviation causes, refer to <xref ref-type="table" rid="table9">Table 9</xref>.</p>
          <table-wrap position="float" id="table9">
            <label>Table 9</label>
            <caption>
              <p>Reasons for large classification errors (≥3 grade deviation) across models. Cases exhibiting high deviations in at least 1 model are included; some cases showed deviations in multiple models.</p>
            </caption>
            <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
              <col width="360"/>
              <col width="90"/>
              <col width="130"/>
              <col width="140"/>
              <col width="140"/>
              <col width="140"/>
              <thead>
                <tr valign="top">
                  <td>Reason</td>
                  <td>Cases, n</td>
                  <td>Qwen3-235B, n</td>
                  <td>Llama-3.3-70B, n</td>
                  <td>GPT-OSS120B, n</td>
                  <td>Ministral-3-8B, n</td>
                </tr>
              </thead>
              <tbody>
                <tr valign="top">
                  <td>Failure to rescue</td>
                  <td>4</td>
                  <td>0</td>
                  <td>4</td>
                  <td>0</td>
                  <td>0</td>
                </tr>
                <tr valign="top">
                  <td>Postoperative ERCP<sup>a</sup></td>
                  <td>7</td>
                  <td>5</td>
                  <td>2</td>
                  <td>6</td>
                  <td>5</td>
                </tr>
                <tr valign="top">
                  <td>Liver failure</td>
                  <td>5</td>
                  <td>4</td>
                  <td>3</td>
                  <td>2</td>
                  <td>4</td>
                </tr>
                <tr valign="top">
                  <td>Renal failure</td>
                  <td>4</td>
                  <td>0</td>
                  <td>1</td>
                  <td>1</td>
                  <td>3</td>
                </tr>
                <tr valign="top">
                  <td>Prolonged ICU<sup>b</sup> monitoring</td>
                  <td>10</td>
                  <td>0</td>
                  <td>0</td>
                  <td>6</td>
                  <td>6</td>
                </tr>
                <tr valign="top">
                  <td>Documentation error</td>
                  <td>2</td>
                  <td>2</td>
                  <td>2</td>
                  <td>1</td>
                  <td>2</td>
                </tr>
                <tr valign="top">
                  <td>Pneumonia with mechanical ventilation</td>
                  <td>2</td>
                  <td>0</td>
                  <td>1</td>
                  <td>1</td>
                  <td>1</td>
                </tr>
                <tr valign="top">
                  <td>Other</td>
                  <td>7</td>
                  <td>1</td>
                  <td>0</td>
                  <td>6</td>
                  <td>2</td>
                </tr>
                <tr valign="top">
                  <td>Total</td>
                  <td>41</td>
                  <td>12</td>
                  <td>13</td>
                  <td>23</td>
                  <td>23</td>
                </tr>
              </tbody>
            </table>
            <table-wrap-foot>
              <fn id="table9fn1">
                <p><sup>a</sup>ERCP: endoscopic retrograde cholangiopancreatography.</p>
              </fn>
              <fn id="table9fn2">
                <p><sup>b</sup>ICU: intensive care unit.</p>
              </fn>
            </table-wrap-foot>
          </table-wrap>
        </sec>
      </sec>
      <sec>
        <title>Majority Vote</title>
        <p>To assess whether aggregating predictions across models could improve classification reliability, a simple majority-vote ensemble was constructed. For each case, the most frequently predicted Clavien-Dindo grade among all open-weight models was selected; ties were resolved by assigning the first occurring grade. Only valid numerical predictions were included in the voting process.</p>
        <p>The majority-vote ensemble exceeds the performance of all individual open-weight and proprietary models in the fine-grained setting (<xref ref-type="table" rid="table10">Table 10</xref>). In the binary grading task, the ensemble reached an accuracy of 0.948, outperforming all individual open-weight models and matching the upper range of the proprietary systems. When expanded to include GPT-5.1 and Gemini 3 Pro Preview, the ensemble across all models achieved 0.794, the highest grade-prediction accuracy observed in the study for fine-grained grading, and maintained a high accuracy for binary grading.</p>
        <table-wrap position="float" id="table10">
          <label>Table 10</label>
          <caption>
            <p>Accuracy of ensemble approaches for fine-grained Clavien-Dindo grade prediction and binary classification. Results shown for open-weight models only and for all models combined.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="370"/>
            <col width="330"/>
            <col width="300"/>
            <thead>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Accuracy for fine-grained grading</td>
                <td>Accuracy for binary grading</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Ensemble open-weights</td>
                <td>0.786</td>
                <td>0.948</td>
              </tr>
              <tr valign="top">
                <td>Ensemble open-weights + proprietary models</td>
                <td>0.794</td>
                <td>0.948</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <p>Additionally, ensemble aggregation reduced large disagreement with expert annotation. Whereas 41 cases exhibited deviations of 3 or more Clavien-Dindo grades in individual model predictions, this number decreased to 14 cases when applying majority voting. This reduction indicates that ensemble-based decision-making effectively mitigates extreme errors by leveraging complementary strengths and partially offsetting individual model biases.</p>
      </sec>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <sec>
        <title>Principal Findings</title>
        <p>In this study, contemporary LLMs demonstrated promising accuracy in classifying postoperative complication severity according to the Clavien-Dindo system using routine discharge letters. Performance was consistently strong across open-weight and proprietary models, with substantial improvement for binary classification of minor versus major complications. No single model uniformly outperformed across all dimensions, highlighting the heterogeneous strengths of different architectures and the value of multimodel strategies. While proprietary models achieved the highest overall accuracy on the balanced 50-case subset, open-weight models demonstrated competitive performance with comparatively lower computational requirements, greater deployment flexibility, and enhanced data privacy compliance, which represent critical advantages for clinical implementation. Given the different dataset sizes used for model evaluation, cross-model comparisons are confined to the balanced 50-case subset. Notably, smaller models such as Ministral-3-8B performed comparably to much larger architectures, suggesting that resource-efficient alternatives may suffice for this task. Ministral-3-8B also exhibited a more conservative response strategy, producing the highest number of “not classifiable” outputs when encountering ambiguous information. While the impact on fine-grained classification was negligible, the effect was more pronounced in binary classification, where the accuracy was reduced by 5% when unclassified cases were treated as prediction errors. To contextualize model performance, interrater reliability between 2 independent clinical annotators yielded a Cohen κ=0.752. Model-to-expert agreement ranged from κ=0.541 to κ=0.663 across open-weight models, falling below but approaching this human benchmark, suggesting that LLM-based classification is operating close to the level of inherent clinician-to-clinician variability in Clavien-Dindo grading. Ensemble aggregation improved performance by leveraging complementary error profiles without requiring additional training or fine-tuning.</p>
      </sec>
      <sec>
        <title>Comparison With Previous Work</title>
        <p>The observed association between Clavien-Dindo grades and length of hospital stay replicates the gradient originally reported by Dindo et al [<xref ref-type="bibr" rid="ref6">6</xref>]. This concordance provides evidence for construct validity of the LLM-based scoring system in the present cohort and indicates that automated classification aligns with established clinical patterns. These findings extend previous work on automated complication grading. Staubli et al [<xref ref-type="bibr" rid="ref30">30</xref>] reported high accuracy (30/31, 96%) using LLMs for Clavien-Dindo classification, but their study relied on clinical scenarios derived from literature rather than original patient documentation. Can et al [<xref ref-type="bibr" rid="ref31">31</xref>] demonstrated that LLMs can extract structured variables from routine interventional oncology reports in hepatobiliary patients, with proprietary models outperforming open-weight alternatives in accuracy and longitudinal consistency. Our study confirms these patterns using real-world discharge letters, which contain unstructured narratives, variable documentation styles, and clinically relevant ambiguities absent from standardized vignettes. Importantly, when evaluated on the same 50-case subset, our results additionally demonstrate that open-weight models can reach competitive performance for specific tasks such as binary complication classification, suggesting that the performance gap between model classes may be task-dependent.</p>
        <p>The Clavien-Dindo classification has become the most widely adopted framework for reporting surgical complications over the past 2 decades. Following amendments in 2009 [<xref ref-type="bibr" rid="ref32">32</xref>], the authors reported 90% interrater agreement. However, empirical research on interrater agreement has remained limited, particularly regarding the effects of clinical experience, time constraints, or documentation quality. Studies in urological and head-and-neck surgery have reported only moderate inter-observer agreement for Clavien-Dindo scoring [<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref34">34</xref>]. Dindo et al [<xref ref-type="bibr" rid="ref6">6</xref>] further demonstrated significant differences in perceived complication severity between patients, nurses, and physicians. The LLMs evaluated in this study achieved higher consistency than reported human interrater reliability in subspecialty settings, suggesting that automated approaches may offer a more reproducible alternative for complication assessment in complex surgical cohorts.</p>
        <p>Logistic regression analysis revealed that increasing document length was associated with lower classification accuracy, suggesting that longer discharge letters introduce narrative complexity rather than discriminative clarity. This underscores the value of concise, structured complication documentation for both automated classification and clinical information transfer. Similarly, determining whether postoperative interventions represent surgical complications or planned disease management proved challenging. Updated consensus guidelines now recommend classifying all postoperative interventions according to Clavien-Dindo regardless of presumed causality, acknowledging that definitive attribution is often clinically impossible [<xref ref-type="bibr" rid="ref35">35</xref>]. The authors proposed organ-specific grading systems to address such liver-specific ambiguities, although this increases complexity and reduces cross-study comparability.</p>
      </sec>
      <sec>
        <title>Clinical Implications</title>
        <p>These findings have direct implications for clinical workflow integration and surgical outcomes research. LLM-based systems could automatically propose Clavien-Dindo grades after discharge documentation is completed, functioning as decision-support tools that reduce documentation burden while preserving physician oversight. In the envisioned deployment scenario, the treating clinician would receive an automated grade recommendation directly upon completion of the discharge letter and could either accept or modify the suggested grade before it is permanently recorded in the hospital information system. Thereby preserving clinical authority while reducing manual abstraction effort and contributing to a continuously growing, structured outcomes database. This approach embeds automated grading as a decision-support artifact that augments rather than replaces clinical judgment, consistent with evidence that machine learning–based clinical decision support systems achieve high clinician acceptance when designed to preserve physician oversight [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>]. To mitigate the risk of high-impact misclassifications in clinical deployment, we propose that document characteristics associated with elevated misclassification risk, such as increased discharge letter length or the presence of ambiguous procedural terminology, should serve as additional flags for prioritized human oversight. For broader adoption in clinical practice, AI-derived classifications will ultimately require structured and interoperable data representations. HL7 FHIR provides a potential framework for integrating such outputs into clinical information systems [<xref ref-type="bibr" rid="ref23">23</xref>]. As implementation aspects were beyond the scope of this study, further considerations are provided in Figure S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
        <p>Beyond real-time use, automated grading is particularly well-suited for retrospective research. Large surgical cohorts can be efficiently annotated when complication grading was not performed systematically at the time of care, addressing a long-standing bottleneck in outcomes research. By bridging unstructured clinical narratives and structured metrics, LLMs offer scalable, reproducible quality assessment that addresses interrater variability and practical constraints limiting broader adoption of systematic complication scoring.</p>
      </sec>
      <sec>
        <title>Limitations</title>
        <p>Despite overall strong performance, a subset of cases demonstrated significant deviations between LLM predictions and expert assessments, reflecting a fundamental tension in the Clavien-Dindo system: classification depends on the therapy required, yet the clinical significance of laboratory abnormalities without intervention remains subject to interpretation, a judgment often not explicitly documented in discharge letters. More broadly, it should be acknowledged that all classifications, both by human annotators and LLMs, are based exclusively on the information documented in discharge letters, which may not fully reflect the true clinical course. Documentation quality, narrative style, and individual reporting practices may influence the assigned grade, and the system therefore assesses documented rather than verified clinical outcomes. While this does not introduce differential bias in the study design of this study, it represents an inherent limitation of documentation-based outcome classification that should be considered when interpreting results in the context of real-world quality monitoring. Future work should explore richer input representations, including operative notes, progress notes, laboratory values, and imaging reports, to assess the added value of longitudinal documentation for complication grading.</p>
        <p>Furthermore, expert annotation was performed by a single senior surgeon, which may introduce individual bias in the assignment of Clavien-Dindo grades. To partially address this limitation, a stratified 10% subset was independently annotated by a second clinician. Complete dual annotation of all 650 cases was not feasible due to the substantial time investment required for expert-level Clavien-Dindo annotation of complex hepatobiliary discharge letters. Disagreements between the 2 annotators were reviewed but did not inform retrospective refinement of the primary annotations, as selectively revising only a subset of the reference standard would have introduced a different form of bias. Formal consensus adjudication was considered but not implemented, as it would have required a third independent expert annotator and represents an avenue for future work. However, interrater agreement was assessed on this subset only and cannot be assumed to be representative. The remaining cases rely solely on single-annotator classification, and observed model-to-expert nonagreement in these cases may therefore partly reflect individual annotator variability rather than true model error. Future studies should incorporate multiple annotators with varying levels of clinical experience to establish a more robust reference standard. In addition, the cases originated from a single care center, reflecting center-specific documentation styles, perioperative pathways, and clinical workflows. Performance may therefore differ in multicenter settings or institutions with alternative reporting standards [<xref ref-type="bibr" rid="ref38">38</xref>]. At the same time, the 14-year observation period introduces the possibility of temporal changes in documentation practice, which may influence the model performance over time. Multicenter validation studies are essential to assess generalizability and transferability across different clinical contexts.</p>
        <p>A further methodological limitation concerns the comparability of open-weight and proprietary model evaluations. Open-weight models were evaluated on the full 650-case dataset, while proprietary models were assessed on a balanced 50-case subset due to cost and access constraints. Although all models were additionally evaluated on the same 50-case subset to enable direct comparison, the difference in evaluation scope limits the interpretability of cross-model conclusions drawn from the full dataset results. All models were evaluated in a zero-shot setting, without task-specific fine-tuning or domain adaptation, thereby limiting performance to intrinsic out-of-the-box reasoning capabilities. Accordingly, the results of this study should be interpreted as a reproducible baseline assessment rather than an optimized deployment scenario. While this choice enables a fair and transparent comparison across models, future work should explore fine-tuning and few-shot prompting, and retrieval-augmented approaches. These methods represent promising strategies for improving performance on challenging grades such as IVa and IVb, where zero-shot performance remained poor across all models. In particular, encoder-only architectures specifically optimized for text classification tasks may represent a more resource-efficient alternative to large generative models and still have high performance. Concurrently, aggregation methods that extend beyond simple majority voting (eg, LLM-as-a-judge frameworks or model councils) have the potential to enhance reliability and clinical alignment by explicitly resolving disagreements, particularly in cases of borderline or rare complication grades. In this context, the robust performance of Ministral-3-8B indicates that further investigation is warranted into SLMs. Encoder-based architectures optimized for text classification may represent a more resource-efficient alternative, with reduced computational cost, inference latency, and ease of integration into clinical IT infrastructures. Nevertheless, the generalizability of these findings is constrained by the size and scope of the dataset. Larger, more diverse, multicenter datasets spanning additional surgical domains are needed to enhance robustness, particularly for rare complication grades. Furthermore, prospective validation in real-world clinical workflows remains a critical next step. Such studies are necessary to evaluate practical utility, assess effects on documentation burden and clinical decision-making, and identify potential biases introduced by automated complication grading systems. Furthermore, although all models were prompted to provide a cited passage as an interpretability basis for their grade assignments, citation accuracy was not formally evaluated and should be considered indicative rather than verified, representing an important direction for future research.</p>
      </sec>
      <sec>
        <title>Conclusion</title>
        <p>This study shows that contemporary LLMs can reliably classify postoperative complications according to the Clavien-Dindo system using routine clinical documentation. Open-weight models offer a particularly attractive trade-off by combining competitive accuracy with substantially lower computational demands, while ensemble strategies further enhance robustness. Notably, the performance of the best open-weight models approached the upper bound of human interrater agreement, suggesting that remaining discrepancies between model and expert annotation may partly reflect the inherent variability of clinical grading rather than model limitations alone. Overall, these results highlight the potential of LLMs as a useful tool for supporting surgical complication assessment at scale.</p>
      </sec>
    </sec>
  </body>
  <back>
    <app-group>
      <supplementary-material id="app1">
        <label>Multimedia Appendix 1</label>
        <p>Supplementary material including all additional text boxes, tables, and figures referenced in the manuscript, providing extended methodological details, model-specific performance metrics, prompt templates, and cohort characteristics.</p>
        <media xlink:href="jmir_v28i1e92407_app1.docx" xlink:title="DOCX File , 2418 KB"/>
      </supplementary-material>
    </app-group>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">CRLM</term>
          <def>
            <p>colorectal liver metastases</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">FHIR</term>
          <def>
            <p>Fast Healthcare Interoperability Resources</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb3">GDPR</term>
          <def>
            <p>General Data Protection Regulation</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb4">HCC</term>
          <def>
            <p>hepatocellular carcinoma</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb5">ICC</term>
          <def>
            <p>intrahepatic cholangiocarcinoma</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb6">ICU</term>
          <def>
            <p>intensive care unit</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb7">LLM</term>
          <def>
            <p>large language model</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb8">LOS</term>
          <def>
            <p>length of stay</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb9">OR</term>
          <def>
            <p>odds ratio</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb10">SLM</term>
          <def>
            <p>small language model</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>We acknowledge the use of ChatGPT, Claude, and Grammarly during the early stages of manuscript drafting to assist in structuring initial ideas and refining wording. All content generated or suggested by these tools was thoroughly reviewed, revised, and edited by the authors, and we take full responsibility for the accuracy and integrity of the final manuscript. The data for this project were provided by the Smart Hospital Information Platform (SHIP), managed by the Data Integration Center at the University Medicine Essen. SHIP serves as a comprehensive digital health platform for integrating data from all major clinical subsystems using a holistic FHIR-based approach. It enables the purification, analysis, distribution, and visualization of clinical data.</p>
    </ack>
    <notes>
      <title>Data Availability</title>
      <p>The dataset used in this study is not publicly available. Individuals or academic organizations interested in utilizing this dataset must submit a detailed request to “Data-Governance@uk-essen.de,” which will be reviewed on a case-by-case basis. We plan to publish the code and make our pipeline available under the repository “UMEssen/CAMEL” on GitHub.</p>
    </notes>
    <notes>
      <title>Funding</title>
      <p>The authors declared no financial support was received for this work.</p>
    </notes>
    <fn-group>
      <fn fn-type="con">
        <p>SW, SMS, and RH contributed to data curation, formal analysis, and writing – original draft. JB, DH, MR, and SMS contributed to data curation. KA and AIY contributed to methodological development. KB contributed to the validation of the statistical analysis. MM and SMS contributed to data interpretation. AÖ, TFU, and UPN contributed medical expertise. FN, UPN, SMS, and RH contributed to conceptualization and supervision. All authors critically revised the manuscript and approved the final version.</p>
      </fn>
      <fn fn-type="conflict">
        <p>None declared.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Shi</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Kinsman</surname>
              <given-names>WC</given-names>
            </name>
            <name name-style="western">
              <surname>Ha</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>GG</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>DI++: A deep learning system for patient condition identification in clinical notes</article-title>
          <source>Artif Intell Med</source>
          <year>2022</year>
          <month>01</month>
          <volume>123</volume>
          <fpage>102224</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/34998515"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.artmed.2021.102224</pub-id>
          <pub-id pub-id-type="medline">34998515</pub-id>
          <pub-id pub-id-type="pii">S0933-3657(21)00217-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC8832473</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Han</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>RF</given-names>
            </name>
            <name name-style="western">
              <surname>Shi</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Richie</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Tseng</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Quan</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Ryan</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Brent</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Tsui</surname>
              <given-names>FR</given-names>
            </name>
          </person-group>
          <article-title>Classifying social determinants of health from unstructured electronic health records using deep learning-based natural language processing</article-title>
          <source>J Biomed Inform</source>
          <year>2022</year>
          <month>03</month>
          <volume>127</volume>
          <fpage>103984</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S1532-0464(21)00313-0"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jbi.2021.103984</pub-id>
          <pub-id pub-id-type="medline">35007754</pub-id>
          <pub-id pub-id-type="pii">S1532-0464(21)00313-0</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Dev</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Zolensky</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Aridi</surname>
              <given-names>HD</given-names>
            </name>
            <name name-style="western">
              <surname>Kelty</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Madison</surname>
              <given-names>MK</given-names>
            </name>
            <name name-style="western">
              <surname>Motaganahalli</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Brooke</surname>
              <given-names>BS</given-names>
            </name>
            <name name-style="western">
              <surname>Dixon</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Boustani</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Ben Miled</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Gonzalez</surname>
              <given-names>AA</given-names>
            </name>
          </person-group>
          <article-title>Use of deep learning to identify peripheral arterial disease cases from narrative clinical notes</article-title>
          <source>J Surg Res</source>
          <year>2024</year>
          <month>11</month>
          <volume>303</volume>
          <fpage>699</fpage>
          <lpage>708</lpage>
          <pub-id pub-id-type="doi">10.1016/j.jss.2024.09.062</pub-id>
          <pub-id pub-id-type="medline">39454287</pub-id>
          <pub-id pub-id-type="pii">S0022-4804(24)00610-3</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Obeid</surname>
              <given-names>JS</given-names>
            </name>
            <name name-style="western">
              <surname>Heider</surname>
              <given-names>PM</given-names>
            </name>
            <name name-style="western">
              <surname>Weeda</surname>
              <given-names>ER</given-names>
            </name>
            <name name-style="western">
              <surname>Matuskowitz</surname>
              <given-names>AJ</given-names>
            </name>
            <name name-style="western">
              <surname>Carr</surname>
              <given-names>CM</given-names>
            </name>
            <name name-style="western">
              <surname>Gagnon</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Crawford</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Meystre</surname>
              <given-names>SM</given-names>
            </name>
          </person-group>
          <article-title>Impact of de-identification on clinical text classification using traditional and deep learning classifiers</article-title>
          <source>Stud Health Technol Inform</source>
          <year>2019</year>
          <month>08</month>
          <day>21</day>
          <volume>264</volume>
          <fpage>283</fpage>
          <lpage>287</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/31437930"/>
          </comment>
          <pub-id pub-id-type="doi">10.3233/SHTI190228</pub-id>
          <pub-id pub-id-type="medline">31437930</pub-id>
          <pub-id pub-id-type="pii">SHTI190228</pub-id>
          <pub-id pub-id-type="pmcid">PMC6779034</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Abbassi</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Pfister</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Lucas</surname>
              <given-names>KL</given-names>
            </name>
            <name name-style="western">
              <surname>Domenghino</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Puhan</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Clavien</surname>
              <given-names>P</given-names>
            </name>
            <collab>Outcome Reporting Group</collab>
          </person-group>
          <article-title>Milestones in surgical complication reporting: Clavien-Dindo classification 20 years and comprehensive complication index 10 years</article-title>
          <source>Ann Surg</source>
          <year>2024</year>
          <volume>280</volume>
          <issue>5</issue>
          <fpage>763</fpage>
          <lpage>771</lpage>
          <pub-id pub-id-type="doi">10.1097/SLA.0000000000006471</pub-id>
          <pub-id pub-id-type="medline">39101214</pub-id>
          <pub-id pub-id-type="pii">00000658-202411000-00010</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Dindo</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Demartines</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Clavien</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>Classification of surgical complications: a new proposal with evaluation in a cohort of 6336 patients and results of a survey</article-title>
          <source>Ann Surg</source>
          <year>2004</year>
          <month>08</month>
          <volume>240</volume>
          <issue>2</issue>
          <fpage>205</fpage>
          <lpage>13</lpage>
          <pub-id pub-id-type="doi">10.1097/01.sla.0000133083.54934.ae</pub-id>
          <pub-id pub-id-type="medline">15273542</pub-id>
          <pub-id pub-id-type="pii">00000658-200408000-00003</pub-id>
          <pub-id pub-id-type="pmcid">PMC1360123</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Van Veen</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Van Uden</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Blankemeier</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Delbrouck</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Aali</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Bluethgen</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Pareek</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Polacin</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Reis</surname>
              <given-names>EP</given-names>
            </name>
            <name name-style="western">
              <surname>Seehofnerová</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Rohatgi</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Hosamani</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Collins</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Ahuja</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Langlotz</surname>
              <given-names>CP</given-names>
            </name>
            <name name-style="western">
              <surname>Hom</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Gatidis</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Pauly</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chaudhari</surname>
              <given-names>AS</given-names>
            </name>
          </person-group>
          <article-title>Adapted large language models can outperform medical experts in clinical text summarization</article-title>
          <source>Nat Med</source>
          <year>2024</year>
          <month>04</month>
          <volume>30</volume>
          <issue>4</issue>
          <fpage>1134</fpage>
          <lpage>1142</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-024-02855-5</pub-id>
          <pub-id pub-id-type="medline">38413730</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-024-02855-5</pub-id>
          <pub-id pub-id-type="pmcid">PMC11479659</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Thirunavukarasu</surname>
              <given-names>AJ</given-names>
            </name>
            <name name-style="western">
              <surname>Ting</surname>
              <given-names>DSJ</given-names>
            </name>
            <name name-style="western">
              <surname>Elangovan</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Gutierrez</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Tan</surname>
              <given-names>TF</given-names>
            </name>
            <name name-style="western">
              <surname>Ting</surname>
              <given-names>DSW</given-names>
            </name>
          </person-group>
          <article-title>Large language models in medicine</article-title>
          <source>Nat Med</source>
          <year>2023</year>
          <month>08</month>
          <volume>29</volume>
          <issue>8</issue>
          <fpage>1930</fpage>
          <lpage>1940</lpage>
          <pub-id pub-id-type="doi">10.1038/s41591-023-02448-8</pub-id>
          <pub-id pub-id-type="medline">37460753</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41591-023-02448-8</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>He</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Mao</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Ruan</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Lan</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Feng</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Cambria</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>A survey of large language models for healthcare: from data, technology, and applications to accountability and ethics</article-title>
          <source>Information Fusion</source>
          <year>2025</year>
          <month>06</month>
          <volume>118</volume>
          <issue>C</issue>
          <fpage>102963</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1016/j.inffus.2025.102963"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.inffus.2025.102963</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Naveed</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Khan</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Qiu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Saqib</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Anwar</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Usman</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Akhtar</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Barnes</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Mian</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>A comprehensive overview of large language models</article-title>
          <source>ACM Trans Intell Syst Technol</source>
          <year>2025</year>
          <month>10</month>
          <volume>16</volume>
          <issue>5</issue>
          <fpage>1</fpage>
          <lpage>72</lpage>
          <pub-id pub-id-type="doi">10.1145/3744746</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Garcia-Carmona</surname>
              <given-names>AM</given-names>
            </name>
            <name name-style="western">
              <surname>Prieto</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Puertas</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Beunza</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Leveraging large language models for accurate retrieval of patient information from medical reports: systematic evaluation study</article-title>
          <source>JMIR AI</source>
          <year>2025</year>
          <month>07</month>
          <day>03</day>
          <volume>4</volume>
          <fpage>e68776</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://ai.jmir.org/2025//e68776/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/68776</pub-id>
          <pub-id pub-id-type="medline">40608403</pub-id>
          <pub-id pub-id-type="pii">v4i1e68776</pub-id>
          <pub-id pub-id-type="pmcid">PMC12271962</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>WX</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Dong</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Hou</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Min</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Du</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Jiang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Ren</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Hu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Nie</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wen</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>A survey of large language models</article-title>
          <source>Front Comput Sci</source>
          <year>2026</year>
          <month>05</month>
          <day>09</day>
          <volume>20</volume>
          <issue>12</issue>
          <fpage>2012627</fpage>
          <pub-id pub-id-type="doi">10.1007/s11704-026-60308-3</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Qu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Yu</surname>
              <given-names>CL</given-names>
            </name>
            <name name-style="western">
              <surname>Pan</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Cheng</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Yuan</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Zheng</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Pang</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Du</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Liang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ma</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ma</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Benetos</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Ma</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Niu</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Que</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Guo</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Gao</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Bai</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Ouyang</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>W</given-names>
            </name>
          </person-group>
          <article-title>MAP-Neo: highly capable and transparent bilingual large language model series</article-title>
          <source>arXiv. Preprint posted online on May 29, 2024</source>
          <pub-id pub-id-type="doi">10.48550/arXiv.2405.19327</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Singhal</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Azizi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Tu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Mahdavi</surname>
              <given-names>SS</given-names>
            </name>
            <name name-style="western">
              <surname>Wei</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Chung</surname>
              <given-names>HW</given-names>
            </name>
            <name name-style="western">
              <surname>Scales</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Tanwani</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Cole-Lewis</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Pfohl</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Payne</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Seneviratne</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Gamble</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Kelly</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Babiker</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Schärli</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Chowdhery</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mansfield</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Demner-Fushman</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Agüera Y Arcas</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Webster</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Corrado</surname>
              <given-names>GS</given-names>
            </name>
            <name name-style="western">
              <surname>Matias</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Chou</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Gottweis</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Tomasev</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Rajkomar</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Barral</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Semturs</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Karthikesalingam</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Natarajan</surname>
              <given-names>V</given-names>
            </name>
          </person-group>
          <article-title>Large language models encode clinical knowledge</article-title>
          <source>Nature</source>
          <year>2023</year>
          <month>08</month>
          <volume>620</volume>
          <issue>7972</issue>
          <fpage>172</fpage>
          <lpage>180</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/37438534"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41586-023-06291-2</pub-id>
          <pub-id pub-id-type="medline">37438534</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41586-023-06291-2</pub-id>
          <pub-id pub-id-type="pmcid">PMC10396962</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Grattafiori</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Dubey</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Jauhri</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Pandey</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Kadian</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Al-Dahle</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Letman</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mathur</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>The Llama 3 herd of models</article-title>
          <source>arXiv. Preprint posted online on July 31, 2024</source>
          <pub-id pub-id-type="doi">10.48550/arXiv.2407.21783</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="web">
          <article-title>Introducing Mistral 3</article-title>
          <source>Mistral AI</source>
          <year>2025</year>
          <access-date>2026-07-18</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://mistral.ai/news/mistral-3/">https://mistral.ai/news/mistral-3/</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bai</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Bai</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Chu</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Cui</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Dang</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Deng</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Fan</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ge</surname>
              <given-names>W</given-names>
            </name>
          </person-group>
          <article-title>Qwen technical report</article-title>
          <source>arXiv. Preprint posted online on September 28, 2023</source>
          <pub-id pub-id-type="doi">10.48550/arXiv.2309.16609</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref18">
        <label>18</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <collab>OpenAI</collab>
          </person-group>
          <source>arXiv. Preprint posted online on August 8, 2025</source>
          <pub-id pub-id-type="doi">10.48550/arXiv.2508.10925</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref19">
        <label>19</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>SH</given-names>
            </name>
            <name name-style="western">
              <surname>Schramm</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Adams</surname>
              <given-names>LC</given-names>
            </name>
            <name name-style="western">
              <surname>Braren</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Bressem</surname>
              <given-names>KK</given-names>
            </name>
            <name name-style="western">
              <surname>Keicher</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Platzek</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Paprottka</surname>
              <given-names>KJ</given-names>
            </name>
            <name name-style="western">
              <surname>Zimmer</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Hedderich</surname>
              <given-names>DM</given-names>
            </name>
            <name name-style="western">
              <surname>Wiestler</surname>
              <given-names>B</given-names>
            </name>
          </person-group>
          <article-title>Benchmarking the diagnostic performance of open source LLMs in 1933 Eurorad case reports</article-title>
          <source>NPJ Digit Med</source>
          <year>2025</year>
          <month>02</month>
          <day>12</day>
          <volume>8</volume>
          <issue>1</issue>
          <fpage>97</fpage>
          <pub-id pub-id-type="doi">10.1038/s41746-025-01488-3</pub-id>
          <pub-id pub-id-type="medline">39934372</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-025-01488-3</pub-id>
          <pub-id pub-id-type="pmcid">PMC11814077</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref20">
        <label>20</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Jiang</surname>
              <given-names>AQ</given-names>
            </name>
            <name name-style="western">
              <surname>Sablayrolles</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mensch</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Bamford</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Chaplot</surname>
              <given-names>DS</given-names>
            </name>
            <name name-style="western">
              <surname>de las Casas</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Bressand</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Lengyel</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Lample</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Saulnier</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Lavaud</surname>
              <given-names>LR</given-names>
            </name>
            <name name-style="western">
              <surname>Lachaux</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Stock</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Scao</surname>
              <given-names>TL</given-names>
            </name>
            <name name-style="western">
              <surname>Lavril</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Lacroix</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Sayed</surname>
              <given-names>W</given-names>
            </name>
          </person-group>
          <article-title>Mistral 7B</article-title>
          <source>arXiv. Preprint posted online on October 10, 2023</source>
          <pub-id pub-id-type="doi">10.48550/arXiv.2310.06825</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref21">
        <label>21</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Touvron</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Lavril</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Izacard</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Martinet</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Lachaux</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Lacroix</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Rozière</surname>
              <given-names>B</given-names>
            </name>
          </person-group>
          <article-title>LLaMA: open and efficient foundation language models</article-title>
          <source>arXiv. Preprint posted online on February 27, 2023</source>
          <pub-id pub-id-type="doi">10.48550/arXiv.2302.13971</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref22">
        <label>22</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Irugalbandara</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Mahendra</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Daynauth</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Arachchige</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Dantanarayana</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Flautner</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Kang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Mars</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Scaling down to scale up: a cost-benefit analysis of replacing OpenAI's LLM with open source SLMs in production</article-title>
          <year>2024</year>
          <conf-name>2024 IEEE International Symposium on Performance Analysis of Systems and Software (ISPASS)</conf-name>
          <conf-date>May 5-7, 2024</conf-date>
          <conf-loc>Indianapolis, IN</conf-loc>
          <fpage>280</fpage>
          <lpage>291</lpage>
          <pub-id pub-id-type="doi">10.1109/ISPASS61541.2024.00034</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref23">
        <label>23</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bender</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Sartipi</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>HL7 FHIR: an agile and RESTful approach to healthcare information exchange</article-title>
          <year>2013</year>
          <month>06</month>
          <conf-name>Proceedings of the 26th IEEE International Symposium on Computer-Based Medical Systems</conf-name>
          <conf-date>June 20-22, 2013</conf-date>
          <conf-loc>Porto, Portugal</conf-loc>
          <fpage>326</fpage>
          <lpage>331</lpage>
          <pub-id pub-id-type="doi">10.1109/cbms.2013.6627810</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref24">
        <label>24</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Zhuang</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Fang</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Gong</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Accuracy of large language models when answering clinical research questions: systematic review and network meta-analysis</article-title>
          <source>J Med Internet Res</source>
          <year>2025</year>
          <month>04</month>
          <day>30</day>
          <volume>27</volume>
          <fpage>e64486</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2025//e64486/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/64486</pub-id>
          <pub-id pub-id-type="medline">40305085</pub-id>
          <pub-id pub-id-type="pii">v27i1e64486</pub-id>
          <pub-id pub-id-type="pmcid">PMC12079073</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref25">
        <label>25</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ronsivalle</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Santonocito</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Cammarata</surname>
              <given-names>U</given-names>
            </name>
            <name name-style="western">
              <surname>Lo Muzio</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Cicciù</surname>
              <given-names>Marco</given-names>
            </name>
          </person-group>
          <article-title>Current applications of chatbots powered by large language models in oral and maxillofacial surgery: a systematic review</article-title>
          <source>Dent J (Basel)</source>
          <year>2025</year>
          <month>06</month>
          <day>11</day>
          <volume>13</volume>
          <issue>6</issue>
          <fpage>261</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.mdpi.com/resolver?pii=dj13060261"/>
          </comment>
          <pub-id pub-id-type="doi">10.3390/dj13060261</pub-id>
          <pub-id pub-id-type="medline">40559164</pub-id>
          <pub-id pub-id-type="pii">dj13060261</pub-id>
          <pub-id pub-id-type="pmcid">PMC12192168</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref26">
        <label>26</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Pan</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Bi</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Wan</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Song</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Fan</surname>
              <given-names>X</given-names>
            </name>
          </person-group>
          <article-title>Evaluating large language models in ophthalmology: systematic review</article-title>
          <source>J Med Internet Res</source>
          <year>2025</year>
          <month>10</month>
          <day>27</day>
          <volume>27</volume>
          <fpage>e76947</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2025//e76947/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/76947</pub-id>
          <pub-id pub-id-type="medline">41144954</pub-id>
          <pub-id pub-id-type="pii">v27i1e76947</pub-id>
          <pub-id pub-id-type="pmcid">PMC12603593</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref27">
        <label>27</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Daccache</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Zako</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Morisson</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Laferrière-Langlois</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>The applications of ChatGPT and other large language models in anesthesiology and critical care: a systematic review</article-title>
          <source>Can J Anaesth</source>
          <year>2025</year>
          <month>06</month>
          <volume>72</volume>
          <issue>6</issue>
          <fpage>904</fpage>
          <lpage>922</lpage>
          <pub-id pub-id-type="doi">10.1007/s12630-025-02973-9</pub-id>
          <pub-id pub-id-type="medline">40524117</pub-id>
          <pub-id pub-id-type="pii">10.1007/s12630-025-02973-9</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref28">
        <label>28</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Arzideh</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Baldini</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Winnekens</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Friedrich</surname>
              <given-names>CM</given-names>
            </name>
            <name name-style="western">
              <surname>Nensa</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Idrissi-Yaghir</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Hosch</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>A Transformer-Based Pipeline for German Clinical Document De-Identification</article-title>
          <source>Appl Clin Inform</source>
          <year>2025</year>
          <month>01</month>
          <volume>16</volume>
          <issue>1</issue>
          <fpage>31</fpage>
          <lpage>43</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://www.thieme-connect.com/DOI/DOI?10.1055/a-2424-1989"/>
          </comment>
          <pub-id pub-id-type="doi">10.1055/a-2424-1989</pub-id>
          <pub-id pub-id-type="medline">39778706</pub-id>
          <pub-id pub-id-type="pmcid">PMC11710903</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref29">
        <label>29</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Elkoushy</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Luz</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Benidir</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Aldousari</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Aprikian</surname>
              <given-names>AG</given-names>
            </name>
            <name name-style="western">
              <surname>Andonian</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Clavien classification in urology: is there concordance among post-graduate trainees and attending urologists?</article-title>
          <source>Can Urol Assoc J</source>
          <year>2013</year>
          <volume>7</volume>
          <issue>5-6</issue>
          <fpage>179</fpage>
          <lpage>84</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/23826044"/>
          </comment>
          <pub-id pub-id-type="doi">10.5489/cuaj.505</pub-id>
          <pub-id pub-id-type="medline">23826044</pub-id>
          <pub-id pub-id-type="pii">cuaj-5-6-179</pub-id>
          <pub-id pub-id-type="pmcid">PMC3699078</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref30">
        <label>30</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Staubli</surname>
              <given-names>SM</given-names>
            </name>
            <name name-style="western">
              <surname>Walker</surname>
              <given-names>HL</given-names>
            </name>
            <name name-style="western">
              <surname>Saner</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Salinas</surname>
              <given-names>CH</given-names>
            </name>
            <name name-style="western">
              <surname>Broering</surname>
              <given-names>DC</given-names>
            </name>
            <name name-style="western">
              <surname>Malagò</surname>
              <given-names>Massimo</given-names>
            </name>
            <name name-style="western">
              <surname>Spiro</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Raptis</surname>
              <given-names>DA</given-names>
            </name>
            <collab>HeALgroup.AI</collab>
          </person-group>
          <article-title>Decoding the Clavien-Dindo Classification: artificial intelligence (AI) as a novel tool to grade postoperative complications</article-title>
          <source>Ann Surg</source>
          <year>2024</year>
          <month>06</month>
          <day>17</day>
          <fpage>273</fpage>
          <pub-id pub-id-type="doi">10.1097/SLA.0000000000006399</pub-id>
          <pub-id pub-id-type="medline">38881457</pub-id>
          <pub-id pub-id-type="pii">00000658-990000000-00938</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref31">
        <label>31</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Can</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Uller</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Kotter</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Vogt</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Doppler</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Brönnimann</surname>
              <given-names>Michael</given-names>
            </name>
            <name name-style="western">
              <surname>Alshinibr</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Elkilany</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Busch</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Kader</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Gassenmaier</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Afat</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Makowski</surname>
              <given-names>MR</given-names>
            </name>
            <name name-style="western">
              <surname>Bressem</surname>
              <given-names>KK</given-names>
            </name>
            <name name-style="western">
              <surname>Adams</surname>
              <given-names>LC</given-names>
            </name>
          </person-group>
          <article-title>Comparative evaluation of proprietary and open-source large language models for systematic multi-source information extraction in interventional oncology</article-title>
          <source>Cardiovasc Intervent Radiol</source>
          <year>2026</year>
          <month>05</month>
          <volume>49</volume>
          <issue>5</issue>
          <fpage>992</fpage>
          <lpage>1004</lpage>
          <pub-id pub-id-type="doi">10.1007/s00270-025-04287-1</pub-id>
          <pub-id pub-id-type="medline">41354880</pub-id>
          <pub-id pub-id-type="pii">10.1007/s00270-025-04287-1</pub-id>
          <pub-id pub-id-type="pmcid">PMC13156197</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref32">
        <label>32</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Clavien</surname>
              <given-names>PA</given-names>
            </name>
            <name name-style="western">
              <surname>Barkun</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>de Oliveira</surname>
              <given-names>ML</given-names>
            </name>
            <name name-style="western">
              <surname>Vauthey</surname>
              <given-names>JN</given-names>
            </name>
            <name name-style="western">
              <surname>Dindo</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Schulick</surname>
              <given-names>RD</given-names>
            </name>
            <name name-style="western">
              <surname>de Santibañes</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Pekolj</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Slankamenac</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Bassi</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Graf</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Vonlanthen</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Padbury</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Cameron</surname>
              <given-names>JL</given-names>
            </name>
            <name name-style="western">
              <surname>Makuuchi</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>The Clavien-Dindo classification of surgical complications: five-year experience</article-title>
          <source>Ann Surg</source>
          <year>2009</year>
          <month>08</month>
          <volume>250</volume>
          <issue>2</issue>
          <fpage>187</fpage>
          <lpage>96</lpage>
          <pub-id pub-id-type="doi">10.1097/SLA.0b013e3181b13ca2</pub-id>
          <pub-id pub-id-type="medline">19638912</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref33">
        <label>33</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Poletajew</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Zapała</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Piotrowicz</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Wołyniec</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Sochaj</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Buraczyński</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Lisiński</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Swiniarski</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Radziszewski</surname>
              <given-names>P</given-names>
            </name>
            <collab>Residents Section of Polish Urological Association</collab>
          </person-group>
          <article-title>Interobserver variability of Clavien-Dindo scoring in urology</article-title>
          <source>Int J Urol</source>
          <year>2014</year>
          <month>12</month>
          <volume>21</volume>
          <issue>12</issue>
          <fpage>1274</fpage>
          <lpage>8</lpage>
          <pub-id pub-id-type="doi">10.1111/iju.12576</pub-id>
          <pub-id pub-id-type="medline">25039893</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref34">
        <label>34</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Stone</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Jiang</surname>
              <given-names>ST</given-names>
            </name>
            <name name-style="western">
              <surname>Stahl</surname>
              <given-names>MC</given-names>
            </name>
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>CJ</given-names>
            </name>
            <name name-style="western">
              <surname>Smith</surname>
              <given-names>RV</given-names>
            </name>
            <name name-style="western">
              <surname>Mehta</surname>
              <given-names>V</given-names>
            </name>
          </person-group>
          <article-title>Development and interrater agreement of a novel classification system combining medical and surgical adverse event reporting</article-title>
          <source>JAMA Otolaryngol Head Neck Surg</source>
          <year>2023</year>
          <month>05</month>
          <day>01</day>
          <volume>149</volume>
          <issue>5</issue>
          <fpage>424</fpage>
          <lpage>429</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/36995708"/>
          </comment>
          <pub-id pub-id-type="doi">10.1001/jamaoto.2023.0169</pub-id>
          <pub-id pub-id-type="medline">36995708</pub-id>
          <pub-id pub-id-type="pii">2802785</pub-id>
          <pub-id pub-id-type="pmcid">PMC10064281</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref35">
        <label>35</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ishii</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Mizuguchi</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Harada</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Ota</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Meguro</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Ueki</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Nishidate</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Okita</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Hirata</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>Comprehensive review of post-liver resection surgical complications and a new universal classification and grading system</article-title>
          <source>World J Hepatol</source>
          <year>2014</year>
          <month>10</month>
          <day>27</day>
          <volume>6</volume>
          <issue>10</issue>
          <fpage>745</fpage>
          <lpage>51</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.wjgnet.com/1948-5182/full/v6/i10/745.htm"/>
          </comment>
          <pub-id pub-id-type="doi">10.4254/wjh.v6.i10.745</pub-id>
          <pub-id pub-id-type="medline">25349645</pub-id>
          <pub-id pub-id-type="pmcid">PMC4209419</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref36">
        <label>36</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Schwartz</surname>
              <given-names>JM</given-names>
            </name>
            <name name-style="western">
              <surname>Moy</surname>
              <given-names>AJ</given-names>
            </name>
            <name name-style="western">
              <surname>Rossetti</surname>
              <given-names>SC</given-names>
            </name>
            <name name-style="western">
              <surname>Elhadad</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Cato</surname>
              <given-names>KD</given-names>
            </name>
          </person-group>
          <article-title>Clinician involvement in research on machine learning-based predictive clinical decision support for the hospital setting: A scoping review</article-title>
          <source>J Am Med Inform Assoc</source>
          <year>2021</year>
          <month>03</month>
          <day>01</day>
          <volume>28</volume>
          <issue>3</issue>
          <fpage>653</fpage>
          <lpage>663</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/33325504"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/jamia/ocaa296</pub-id>
          <pub-id pub-id-type="medline">33325504</pub-id>
          <pub-id pub-id-type="pii">6039107</pub-id>
          <pub-id pub-id-type="pmcid">PMC7936403</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref37">
        <label>37</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Berge</surname>
              <given-names>GT</given-names>
            </name>
            <name name-style="western">
              <surname>Granmo</surname>
              <given-names>OC</given-names>
            </name>
            <name name-style="western">
              <surname>Tveit</surname>
              <given-names>TO</given-names>
            </name>
            <name name-style="western">
              <surname>Munkvold</surname>
              <given-names>BE</given-names>
            </name>
            <name name-style="western">
              <surname>Ruthjersen</surname>
              <given-names>AL</given-names>
            </name>
            <name name-style="western">
              <surname>Sharma</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Machine learning-driven clinical decision support system for concept-based searching: a field trial in a Norwegian hospital</article-title>
          <source>BMC Med Inform Decis Mak</source>
          <year>2023</year>
          <month>01</month>
          <day>10</day>
          <volume>23</volume>
          <issue>1</issue>
          <fpage>5</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmedinformdecismak.biomedcentral.com/articles/10.1186/s12911-023-02101-x"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12911-023-02101-x</pub-id>
          <pub-id pub-id-type="medline">36627624</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12911-023-02101-x</pub-id>
          <pub-id pub-id-type="pmcid">PMC9832658</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref38">
        <label>38</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Lyu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Bian</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Hogan</surname>
              <given-names>WR</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>A study of deep learning methods for de-identification of clinical notes in cross-institute settings</article-title>
          <source>BMC Med Inform Decis Mak</source>
          <year>2019</year>
          <month>12</month>
          <day>05</day>
          <volume>19</volume>
          <issue>Suppl 5</issue>
          <fpage>232</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmedinformdecismak.biomedcentral.com/articles/10.1186/s12911-019-0935-4"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12911-019-0935-4</pub-id>
          <pub-id pub-id-type="medline">31801524</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12911-019-0935-4</pub-id>
          <pub-id pub-id-type="pmcid">PMC6894104</pub-id>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
