<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="2.0">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">JMIR</journal-id>
      <journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id>
      <journal-title>Journal of Medical Internet Research</journal-title>
      <issn pub-type="epub">1438-8871</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v28i1e86202</article-id>
      <article-id pub-id-type="pmid">42748424</article-id>
      <article-id pub-id-type="doi">10.2196/86202</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Original Paper</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Original Paper</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Dynamic Prediction of Day Mortality in Patients With Trauma Using a Hybrid Neural Network Model: Model Development and Evaluation Study</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>Coristine</surname>
            <given-names>Andrew</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Wiraguna</surname>
            <given-names>Rayie</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Amakye</surname>
            <given-names>Felix </given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Moglia</surname>
            <given-names>Victoria</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author">
          <name name-style="western">
            <surname>Millarch</surname>
            <given-names>Andreas Skov</given-names>
          </name>
          <degrees>Msc, PhD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0002-1261-9866</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author">
          <name name-style="western">
            <surname>Kaafarani</surname>
            <given-names>Haytham</given-names>
          </name>
          <degrees>MD, MPH</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <xref rid="aff3" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-4682-9135</ext-link>
        </contrib>
        <contrib id="contrib3" contrib-type="author">
          <name name-style="western">
            <surname>Chamseddine</surname>
            <given-names>Ibrahim</given-names>
          </name>
          <degrees>Phd</degrees>
          <xref rid="aff3" ref-type="aff">3</xref>
          <xref rid="aff4" ref-type="aff">4</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-7005-8692</ext-link>
        </contrib>
        <contrib id="contrib4" contrib-type="author">
          <name name-style="western">
            <surname>Folke</surname>
            <given-names>Fredrik</given-names>
          </name>
          <degrees>MD, PhD</degrees>
          <xref rid="aff5" ref-type="aff">5</xref>
          <xref rid="aff6" ref-type="aff">6</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-2284-7857</ext-link>
        </contrib>
        <contrib id="contrib5" contrib-type="author">
          <name name-style="western">
            <surname>Rudolph</surname>
            <given-names>Søren Steeman</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-7846-4612</ext-link>
        </contrib>
        <contrib id="contrib6" contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Sillesen</surname>
            <given-names>Martin</given-names>
          </name>
          <degrees>MD, PhD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <address>
            <institution>Rigshospitalet</institution>
            <addr-line>blegdamsvej 9</addr-line>
            <addr-line>Copenhagen, 2100</addr-line>
            <country>Denmark</country>
            <phone>45 35453545</phone>
            <email>martin.sillesen@regionh.dk</email>
          </address>
          <xref rid="aff6" ref-type="aff">6</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-9494-5475</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff1">
        <label>1</label>
        <institution>Rigshospitalet</institution>
        <addr-line>Copenhagen</addr-line>
        <country>Denmark</country>
      </aff>
      <aff id="aff2">
        <label>2</label>
        <institution>Division of Trauma, Emergency Surgery and surgical critical care</institution>
        <institution>Mass General Brigham</institution>
        <addr-line>Boston, MA</addr-line>
        <country>United States</country>
      </aff>
      <aff id="aff3">
        <label>3</label>
        <institution>Harvard University</institution>
        <addr-line>boston, MA</addr-line>
        <country>United States</country>
      </aff>
      <aff id="aff4">
        <label>4</label>
        <institution>Department of Radiation Oncology</institution>
        <institution>Mass General Brigham</institution>
        <addr-line>Boston, MA</addr-line>
        <country>United States</country>
      </aff>
      <aff id="aff5">
        <label>5</label>
        <institution>Gentofte Hospital</institution>
        <addr-line>Gentofte</addr-line>
        <country>Denmark</country>
      </aff>
      <aff id="aff6">
        <label>6</label>
        <institution>University of Copenhagen</institution>
        <addr-line>Copenhagen, Capital Region</addr-line>
        <country>Denmark</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Martin Sillesen <email>martin.sillesen@regionh.dk</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>16</day>
        <month>9</month>
        <year>2026</year>
      </pub-date>
      <volume>28</volume>
      <elocation-id>e86202</elocation-id>
      <history>
        <date date-type="received">
          <day>20</day>
          <month>10</month>
          <year>2025</year>
        </date>
        <date date-type="rev-request">
          <day>9</day>
          <month>2</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>8</day>
          <month>5</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©Andreas Skov Millarch, Haytham Kaafarani, Ibrahim Chamseddine, Fredrik Folke, Søren Steeman Rudolph, Martin Sillesen. Originally published in the Journal of Medical Internet Research (https://www.jmir.org), 16.09.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on https://www.jmir.org/, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://www.jmir.org/2026/1/e86202" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>The trajectory of a patient with trauma is often complex and nonlinear. Real-time estimation of the mortality risk from prehospital care to discharge is critical for point-of-care decision-making and for benchmarking the quality of care. Conventional risk assessment systems in trauma are simple and data-sparse, leaving potential for harvesting available data for personalized risk assessments accounting for developing patient states.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>This study aims to create an AI risk prediction model for 30-day mortality in patients with trauma capable of performing predictions at any time point through treatment phases from prehospital to discharge.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>Data on pre- and in-hospital care of patients with trauma treated in Denmark, Capital Region between 2017 and 2024 were used. Demographic and comorbidity-specific variables were structured as tabular data. Temporal data including vitals, laboratory test results, and medications were structured as sequences with temporally determined dynamic bin sizes. Trajectories were censored before the outcome event. Across continuous input types, scaling and normalization parameters were fitted on observed values only, and missing values were handled through zero-imputation with auxiliary indicator channels enabling the model to distinguish observed from absent measurements. Missing values across categorical input types were mapped to a reserved token with a learned embedding. The model architecture combines both tabular and sequential input data. A unified input is processed by multiple layers of transformer encoders before being passed into a neural network for binary classification. Model performance was assessed using area under the receiver operating characteristic curve (AUROC) and area under the precision-recall curve (AUPRC) on a holdout dataset comprising 20% of the total data.</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>A total of 9496 patients were included. The model achieved an AUROC of 0.962 (95% CI 0.930-0.994) and an AUPRC of 0.655 (95% CI 0.545-0.765) predicting 30-day mortality in the holdout evaluation set using full-length trajectories, comprising 1829 patients. In active-cohort evaluation, the model achieved an AUROC of 0.905 (95% CI 0.856-0.953) at 1 hour from first patient contact, demonstrating early discriminative capability from the prehospital phase onward. The model outperformed both Revised Trauma Score (mean ΔAUROC +0.297; 50/54 time points significant) and Trauma and Injury Severity Score (mean ΔAUROC +0.169; 46/54 time points significant).</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>We designed a dynamic, automated model that allows risk prediction at any point in time during nonlinear trajectory of the patient with trauma. This study demonstrates that a hybrid neural network model, trained on automatically extracted electronic health record data, can predict 30-day all-cause mortality in patients with trauma with strong discrimination across the care trajectory. The model could be used as decision support for triage, patient deterioration alerts, bedside decision-making, and family counseling. These findings support the feasibility of sequential modeling for trauma risk prediction.</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>dynamic</kwd>
        <kwd>electronic health records</kwd>
        <kwd>hybrid neural network</kwd>
        <kwd>machine learning risk prediction</kwd>
        <kwd>patient trajectory modeling</kwd>
        <kwd>temporal validation</kwd>
        <kwd>transformer</kwd>
        <kwd>trauma risk assessment</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <p>The trajectory of a patient with trauma is often complex and nonlinear, creating a highly dynamic data flow that has the potential to predict clinically relevant outcomes [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. This data flow starts from the prehospital setting, continues through to hospital treatment and discharge, encompassing a wide range of parameters, including physiological measurements, clinical assessments, treatment interventions, and patient responses. Real-time estimation of the mortality risk from prehospital care to discharge is critical for decision-making and benchmarking the quality of care. First, it informs crucial decision-making processes, allowing health care providers to adapt treatment strategies promptly and allocate resources efficiently. Second, it serves as a valuable tool for benchmarking the quality of care across different trauma centers and systems. As health care systems increasingly leverage big data and machine learning (ML) technologies, the integration of dynamic risk assessment models into clinical practice holds promise for significantly improving patient outcomes in trauma care [<xref ref-type="bibr" rid="ref3">3</xref>].</p>
      <p>Conventionally, mortality risk estimation in trauma relies only on summarized information about the patient state at a certain time point, for example, the first value of systolic blood pressure, or by aggregation, such as the lowest systolic blood pressure. A prime example is the widely used Trauma and Injury Severity Score (TRISS) [<xref ref-type="bibr" rid="ref4">4</xref>], which is calculated by trauma type (blunt or penetrating), age group, Revised Trauma Score (RTS) [<xref ref-type="bibr" rid="ref5">5</xref>], and Injury Severity Score (ISS) [<xref ref-type="bibr" rid="ref6">6</xref>]. The RTS, a component of TRISS, is calculated using the initial values of Glasgow Coma Score (GCS) [<xref ref-type="bibr" rid="ref7">7</xref>], systolic blood pressure, and respiratory rate. While these established scoring systems have demonstrated clinical use, they have significant limitations. By reducing complex time-series data to single values, they inherently discard a wealth of potentially valuable information about the patient’s evolving condition. This approach fails to capture the dynamic nature of trauma progression and response to treatment, potentially overlooking critical trends and patterns that could inform more accurate risk assessments and guide clinical decision-making. As health care technology advances, there is a growing opportunity to leverage the continuous stream of patient data being collected, potentially enhancing the precision and timeliness of mortality risk predictions in trauma care.</p>
      <p>ML models offer significant advantages in predicting trauma outcomes, particularly in their ability to handle complex data relationships and numerous input features. These models can capture nonlinear relationships between variables that are often challenging to detect using classical statistical approaches. This capability is especially valuable in trauma care, where patient outcomes may depend on intricate interactions between multiple factors. Previous studies have demonstrated superior predictive performance using ML compared to TRISS in predicting mortality [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>] and the potential to predict posttrauma complications [<xref ref-type="bibr" rid="ref10">10</xref>]. In contrast to most previous studies, which use static information, the model architecture used in this study combines tabular and time-series data through a sequential data fusion approach. This adds a sense of timing and context to the model, which currently can be harvested using attention, enhancing its ability to capture dynamic patterns and relationships [<xref ref-type="bibr" rid="ref11">11</xref>]. Using a modeling approach that can capture intricate interactions between multiple factors with a sense of timing and context of changes in treatment and patient states may improve predictive performance.</p>
      <p>The overall concept is depicted in <xref rid="figure1" ref-type="fig">Figure 1</xref>. Using data automatically extracted from prehospital and in-hospital electronic health record (EHR) for ease of use in an acute clinical setting, we aimed to create a dynamic AI risk prediction model for 30-day mortality in patients with trauma, capable of performing dynamic predictions at any time point, recalculating predictions at each timestep, through treatment phases from prehospital to discharge fully using the array of data generated as patients advance through treatment trajectories. We use the term “dynamic” to refer to this sequential updating of predictions with accumulating input data. In the framing taxonomy of Lauritsen et al [<xref ref-type="bibr" rid="ref12">12</xref>], the model is left-aligned with a fixed prediction window and a variable-length observation window that expands as clinical data accumulates.</p>
      <fig id="figure1" position="float">
        <label>Figure 1</label>
        <caption>
          <p>An example of a patient trajectory consisting of several phases and different data streams feeding into our model as they are created. Thus, displaying the clinical evolution of a patient and the dynamic development in mortality risk estimation over time from prehospital till discharge. BP: blood pressure; DVT: deep venous thrombosis; IV: intravenous; RSI: rapid sequence induction; WBC: white blood cell count.</p>
        </caption>
        <graphic xlink:href="jmir_v28i1e86202_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
      </fig>
      <p>Potential clinical applications of the proposed model include triaging, urgency awareness in the trauma bay, decision support for treatment strategy in the operative setting, structured communication during transitions between treatment phases, patient family counseling, and clinical quality work.</p>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <sec>
        <title>Ethics Statement</title>
        <p>The study was approved by the Danish Patient Safety Board (Styrelsen for Patientsikkerhed, approval #31-1521), and the Danish Capital Regions Data Safety Board (Videncenter for Dataanmeldelser, approval #P-2020-180). In compliance with Danish legislation on health care research, patient consent was not required due to the retrospective and deidentified nature of the dataset and thus not obtained.</p>
      </sec>
      <sec>
        <title>Data Sources and Inclusion</title>
        <p>The study used retrospective data from pre- and in-hospital EHRs from patients with trauma treated at any of the 6 acute care hospitals in the Capital Region of Denmark from 2017 to June 2024. This region covers 1.9 million inhabitants, where health care (including acute care services) is offered free of charge for all citizens. Prehospital triage criteria to either a local acute care hospital or the region’s single level I trauma center (Rigshospitalet) are based on clinical guidelines (Figure S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p>
        <p>Prehospital EHR data were obtained through the Emergency Medical Services, Capital Region of Denmark. In-hospital EHR data were provided through the Capital Region Research Data Analysis Platform.</p>
        <p>Patients were identified by treatment through an in-hospital EHR trauma call registration (Danish procedure code BWST1F). We included patients of all ages who were present in both the pre- and in-hospital EHR with any type of trauma and mechanism of injury. Patients declared dead on scene were excluded from the dataset. Of note, trauma team activation criteria follow local hospital guidelines and are therefore not standardized.</p>
      </sec>
      <sec>
        <title>Outcome</title>
        <p>The model predicts 30-day all-cause mortality calculated from the trauma date. The date of death was obtained from the EHR, which is updated with information on both in- and out-of-hospital deaths using the Danish Central Person Registry. As such, the EHR data also provide information on postdischarge mortality.</p>
        <p>To prevent data leakage from perimortem clinical patterns (eg, terminal vital sign deterioration or treatment withdrawal), the trajectory end time of deceased patients was truncated prior to time binning and feature extraction. A proportional masking strategy was applied, stratified by trajectory duration: the final 10% was removed for short trajectories (≤6 hours), 5% for medium trajectories (6-72 hours), and 2% for long trajectories (&#62;72 hours). This ensures more aggressive masking for shorter stays, where perimortem patterns constitute a larger fraction of the observed data. A minimum postmasking trajectory duration of 30 minutes was enforced. Since masking is applied before all downstream processing, subsequent feature aggregation and normalization only reflect the truncated observation window. Surviving patients are unaffected.</p>
      </sec>
      <sec>
        <title>Dataset Construction</title>
        <p>Trauma cases were identified as defined above (all patients meeting trauma team activation criteria at any acute care hospital in the Capital Region of Denmark). Demographics and comorbidity variables were structured as tabular data and obtained from the EHR. Vital signs, laboratory test results, and medication were structured as sequential time-series data.</p>
        <p>Prehospital data encompassed continuous measurements of vital signs, that is, blood pressure, heart rate, and oxygen saturation, and GCS and airway, breathing, circulation, and disability (ABCD) assessments. In-hospital EHR data consisted of vital signs; GCS; medication groupings (ie, opioids, antibiotics, neurological drugs, antithrombotic drugs, cardiovascular drugs, infusions, anesthetics, diuretics, insulin, hemostatics, and blood transfusions); laboratory results (ie, lactate, base excess, hemoglobin, leukocyte count, and thromboelastogram [Haemonetics Mount Juliet] measures); and admission, discharge, and transfer (ADT) events grouped as operation room, intensive care unit (ICU), and inpatient ward. The underlying Anatomical Therapeutic Chemical (ATC) codes for each medication group are presented in Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Laboratory test groupings and underlying variables are described in Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
        <p>The trajectory beginning was marked using the first available time stamp from the prehospital EHR for each trauma case. The in-hospital starting and ending time points were marked for each case using in-hospital data on ADT events. Using temporally determined binning frequencies for the data segments with increasing bin size over time up to 30 days allowed using a higher granularity of data at the beginning of patient trajectories, where the patient state is also more dynamic, while allowing longer sequences within model architectural limits. Binning frequencies for each segment are presented in Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
        <p>The segments were used to map vital signs, laboratory test results, ISS, and medications across the entire patient trajectory when available. Vital signs, GCS, and laboratory tests were represented at their absolute values, and medication groups were multihot-encoded in each segment. Vital signs were summarized in each segment as the mean and SD of the binned values. Laboratory tests were summarized using the maximum value in each segment.</p>
        <p>Diagnosis codes, based on the <italic>ICD</italic> (<italic>International Classification of Diseases</italic>), prior to the trauma event were used to calculate the Elixhauser Comorbidity Index (ECI) [<xref ref-type="bibr" rid="ref13">13</xref>] using the comorbidity R package [<xref ref-type="bibr" rid="ref14">14</xref>]. Patients without prior diagnosis records were assigned an ECI of zero, as absence from the diagnosis registry was interpreted as no documented comorbidities rather than a missing value. Diagnosis codes registered during the hospital stay were used for estimating ISS using ICDPIC-R [<xref ref-type="bibr" rid="ref15">15</xref>], as previously demonstrated to be feasible in a Danish population with trauma [<xref ref-type="bibr" rid="ref16">16</xref>]. Moreover, ISS registrations were extracted from free-text clinical notes using regular expression pattern matching. To prevent data leakage, ISS was entered into the model only when available, based on the latest relevant diagnosis registration date or note last-edited time stamp, respectively. This ensures the model has access only to information that would be available at each prediction time point in a prospective setting, at the cost of potential input lag relative to the true time of clinical assessment.</p>
        <p>Initial prehospital assessments of ABCD [<xref ref-type="bibr" rid="ref17">17</xref>] were also treated as tabular data using the predetermined values available to emergency service providers nationally on their EHR interface. The classifications are presented in Table S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. In the case of level of affectedness, these categories are not objectively defined but are based on the judgment of the prehospital provider using a mix of established reference ranges, patient-specific factors, and clinical context to decide whether a particular category is abnormal. Therefore, this model input feature should be regarded as a proxy for the clinician’s level of concern regarding each of the ABCD categories.</p>
      </sec>
      <sec>
        <title>Preprocessing</title>
        <p>The complete trauma dataset was split into a training dataset and a holdout dataset. The holdout dataset was split temporally, comprising the newest 20 percent of the complete data, and was only used after model training was completed for evaluation purposes to ensure proper model generalization. All normalization statistics, both tabular and time series, were computed on the training set and applied to the holdout set without refitting, preventing data leakage.</p>
        <p>For the tabular dataset, continuous features were normalized using an adaptive scaling approach where the transformation method (standard, robust, quantile, or power transform) was selected per variable based on skewness and kurtosis. Age and weight used standard scaling while ECI and height used quantile scaling. For continuous static variables with missing values (height and weight), scaling parameters were fitted on observed values only, excluding missing entries. After scaling, remaining missing values were filled with zero in the standardized space, and a binary indicator variable was created per variable, enabling the model to distinguish observed measurements from absent ones. Categorical features were integer-encoded with a per-variable vocabulary, where missing or unknown values were mapped to a reserved index-zero token with a learned embedding, enabling the model to learn a dedicated representation for missingness without requiring a separate indicator variable.</p>
        <p>Time-series data exhibit 2 forms of absence: <italic>missingness</italic> (unrecorded measurements within a patient’s trajectory) and <italic>padding</italic> (positions beyond the trajectory end point). We distinguish these explicitly: trajectory lengths are derived from the data, and a binary padding mask identifies positions beyond each patient’s trajectory boundary.</p>
        <p>Per-channel normalization (adaptive <italic>z</italic>-score, robust, or quantile transform selected per channel based on skewness and kurtosis) is fitted exclusively on observed values, excluding both missing and padding positions. Following normalization, measured values are approximately standard normal, while both missing and padding positions are imputed as zero. This zero-imputation scheme leverages the postnormalization distribution: since measured values are centered at zero with unit variance, the zero token represents the channel-specific population means, providing an uninformative but numerically stable default. An auxiliary channel for binary data presence indication is appended to the input tensor and excluded from normalization, enabling the model to distinguish observed measurements from imputed zeros. For categorical time series, a multi–hot count encoding is applied per time bin, where absence of a category is naturally represented as zero counts, requiring no separate imputation.</p>
        <p>Per-channel normalization is fitted exclusively on observed values, excluding both missing and padding positions. Categorical time series (medications, procedures, and clinical events) use multi–hot count encoding per time bin, where absence is naturally represented as zero counts.</p>
      </sec>
      <sec>
        <title>Training Process</title>
        <p>The complete dataset was temporally split into a training dataset (80%) and a holdout dataset (most recent 20%), with the holdout used only for final evaluation. The training dataset, hereafter referred to as the development dataset, was further split into 80% training and 20% internal validation folds during cross-validation [<xref ref-type="bibr" rid="ref18">18</xref>] using 5 splits (illustrated in Figure S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). For 50 trials, the folds were randomized but stratified on outcome prevalence. Random rather than temporal fold assignment was chosen so that each fold’s training partition spans the full temporal range of the development dataset, matching the conditions under which the final model is trained on the full development dataset. During this process (<xref rid="figure2" ref-type="fig">Figure 2</xref>), a range of hyperparameters was optimized against maximizing the sum of area under the precision-recall curve (AUPRC) across all folds. Using the optimized hyperparameters, the final model was trained on the full development dataset before evaluation on the temporally distinct holdout set.</p>
        <fig id="figure2" position="float">
          <label>Figure 2</label>
          <caption>
            <p>The overall training process displaying how optimal hyperparameters and thresholds were determined through hyperparameter sweeping over 50 trials each of 5-fold cross-validation, optimizing for the sum of area under the precision-recall curve for the internal folds and finally evaluating performance on the holdout dataset using receiver operating characteristic (ROC), precision-recall curve (PR), and confusion matrices. Each trial optimized parameters including: batch size (bs), residual dropout fraction (res_dropout), the model head dropout fraction (head_dropout), number of transformer encoder layers (n_layers), number of attention heads (n_heads), and model dimension (d_model).</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e86202_fig2.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Model Architecture</title>
        <p>Our model used a combination of static contextual features and sequential input data. We expanded on the timeseries-tabular-transformer [<xref ref-type="bibr" rid="ref19">19</xref>], an adaptation of the TabTransformer [<xref ref-type="bibr" rid="ref20">20</xref>] architecture designed for time-series and tabular data fusion. This hybrid neural network approach capitalizes on the attention mechanism’s strengths for processing sequential data while incorporating tabular patient information to contextualize the representation.</p>
        <p>The transformer encoder applies causal masking, ensuring each temporal position attends only to itself and preceding timesteps, preventing information leakage from future observations. A shared-weight per-timestep prediction head produces independent mortality risk estimates at each temporal position, enabling continuous risk monitoring throughout the patient trajectory. Attention masking is used to ignore end-of-sequence padding.</p>
        <p>The model architecture is elaborated in Appendix S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Implementations are listed in Appendix S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
      </sec>
      <sec>
        <title>Performance Evaluation Metrics</title>
        <p>To evaluate performance, we used a receiver operating characteristic (ROC) curve and a precision-recall (PR) curve. The ROC curve was summarized with the area under the receiver operating characteristic curve (AUROC), and for PR we calculated the AUPRC. CIs were calculated by the DeLong method for AUROC and bootstrapping for AUPRC.</p>
        <p>Following the framing taxonomy for ML risk prediction proposed by Lauritsen et al [<xref ref-type="bibr" rid="ref12">12</xref>], the model uses a left-aligned structure with a fixed prediction window (30-day all-cause mortality from first contact with health care providers) and a variable-length observation window that expands as clinical data accumulates The predicted outcome is anchored at trajectory start; consequently, the residual prediction window decreases as the trajectory progresses, though the outcome definition itself remains fixed. We use the term “dynamic” throughout this work to refer to this sequential updating of predictions with accumulating data, rather than to a time-varying outcome definition. This framing has direct implications for evaluation, as described below.</p>
        <p>Dynamic risk prediction performance was evaluated by masking sequences continuously at each step, equivalent to binning frequencies, to plot AUROC and AUPRC as a function of time. Two complementary evaluation strategies were used to characterize model performance under different assumptions about the evaluable population: population-level evaluation and active-cohort evaluation.</p>
        <p>For population-level evaluation, all holdout patients were retained at every evaluation time point regardless of trajectory length. For patients whose hospital trajectory ended before a given evaluation horizon (due to death or discharge), predictions were generated using all available data up to their end point. This approach maintains a fixed cohort across time points, enabling direct comparison of discriminative performance over time without compositional shifts in the denominator population. However, at later evaluation horizons, predictions for short-trajectory patients reflect less temporal information than the horizon implies, and the correlation between trajectory length and outcome introduces informative censoring.</p>
        <p>For active-cohort evaluation, each evaluation time point <italic>t</italic> includes only patients with trajectories extending beyond <italic>t</italic>. This ensures that every evaluated prediction is informed by data spanning the full horizon, yielding a clinically interpretable assessment: given a patient still under observation at time <italic>t</italic>, how well does the model discriminate? The trade-off is survivorship bias since the cohort at later time points is progressively enriched for longer-stay patients and decreasing sample size, which widens CI. Sample sizes are reported at each time point for the active-cohort analysis. We present both strategies jointly to allow performance assessment under fixed-population and conditional-on-observation assumptions, respectively.</p>
        <p>Additionally, we evaluated the model’s ability to identify high-risk patients using a percentile-based ranking approach. At each time point, predictions were ranked, and the top 5%, 10%, 15%, 20%, and 25% were classified as positive. This approach relies solely on the model’s discrimination (rank-ordering of patients) rather than on absolute predicted probabilities, making it robust to miscalibration. Sensitivity was calculated at each percentile threshold and plotted over time with CIs, illustrating the trade-off between case capture and tolerance to false positives across clinically relevant operating points.</p>
        <p>To further assess the reliability of the model’s probability estimates, and following recommendations for comprehensive evaluation of predictive AI models in medicine [<xref ref-type="bibr" rid="ref21">21</xref>], we created calibration plots, tested for optimal post hoc calibration method through expected calibration error (ECE) metric, conducted post calibration net benefit analysis and calculated the Brier score [<xref ref-type="bibr" rid="ref22">22</xref>] at multiple time frames using active-cohort evaluation. Furthermore, we optimized classification thresholds using the <italic>F<sub>β</sub></italic> score, allowing us to balance precision and sensitivity according to the specific applications of our model. We applied <italic>β</italic>=5 for an analysis prioritizing sensitivity and <italic>β</italic>=1 for the harmonic mean between the 2 metrics. These optimized thresholds were then used to construct confusion matrices, which were used for calculating sensitivity, specificity, and positive predictive value (PPV) at a specific threshold.</p>
        <p>
          <bold>Comparison with trauma risk scores</bold>
        </p>
        <p>To contextualize the model’s discriminative performance, we benchmarked against RTS and TRISS. RTS was computed from the first recorded GCS, systolic blood pressure, and respiratory rate using the standard coded-value formulation with Champion coefficients [<xref ref-type="bibr" rid="ref5">5</xref>]. TRISS was computed from RTS, ISS, age (dichotomized at 55 years), and injury mechanism (blunt vs penetrating) using the Major Trauma Outcome Study logistic regression coefficients. Injury mechanism was determined from Abbreviated Injury Scale region codes.</p>
        <p>At each evaluation time point, the model and each score were evaluated on identical patient subsets, restricted to patients with active trajectories for whom the respective score was computable. Coverage varied across scores due to differing input requirements; patients with missing score components were excluded from the respective paired comparison but retained in the model’s primary evaluation. As the traditional scores are time-invariant, computed once from initial presentation data, divergence from the model’s trajectory over time reflects the incremental value of accumulating longitudinal information. Statistical significance of AUROC differences was assessed using paired DeLong tests [<xref ref-type="bibr" rid="ref23">23</xref>] at each time point, with the Benjamini-Hochberg procedure [<xref ref-type="bibr" rid="ref24">24</xref>] applied to control the false discovery rate (FDR) at 5%.</p>
      </sec>
      <sec>
        <title>Model Behavior Analysis</title>
        <p>To interpret the model’s predictions, we used Shapley additive explanations (SHAP) [<xref ref-type="bibr" rid="ref25">25</xref>] using the GradientExplainer algorithm, which approximates Shapley values via expected gradients and is well-suited to deep learning architectures [<xref ref-type="bibr" rid="ref26">26</xref>]. SHA<italic>P</italic> values were computed separately for each input modality: continuous time-series channels, categorical time, static categorical features, and static continuous features.</p>
        <p>For categorical time series, we computed SHA<italic>P</italic> values on the raw multihot representation rather than on pre-embedded vectors, yielding per-category attributions that identify which specific medications or procedures contribute to the prediction at each timestep. Static categorical features were pre-embedded before SHAP computation, with attributions averaged across embedding dimensions to produce a single importance score per feature. A background dataset of 1000 randomly sampled training patients served as the reference distribution for the GradientExplainer.</p>
        <p>SHA<italic>P</italic> values beyond each patient’s actual trajectory length were zeroed to prevent attribution of importance to padding positions. We computed mean absolute SHA<italic>P</italic> values across holdout patients to identify globally important features and their temporal dynamics. For all time-series modalities, per-measurement importance was computed by averaging absolute SHA<italic>P</italic> values across measured (nonzero) positions only, ensuring comparable density-normalized importance estimates across channels with different sparsity profiles. For temporal analysis, we additionally performed time frame–specific SHAP computation at progressively later clinical time points, censoring input data beyond each time point and recomputing SHA<italic>P</italic> values to assess how feature importance evolves as more clinical information becomes available.</p>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <sec>
        <title>Overview</title>
        <p>The study included 9496 trauma cases with a mean patient age of 42.2 (SD 22.2) years; 66.4% (n=6308) were male. The mean ECI was 0.7 (SD 3.2), the ISS mean was 6.6 (SD 10.6), and the GCS median was 15.0 (IQR 13.0-15.0). The 30-day mortality rate was 3.92% (372/9496). Sequential data encompassed 8,357,172 vital sign measurements, 2,424,172 medication administrations, and 3,294,268 laboratory test results over a mean patient trajectory of 7.6 (SD 21.5) days. Patient characteristics are available in <xref ref-type="table" rid="table1">Table 1</xref>.</p>
        <table-wrap position="float" id="table1">
          <label>Table 1</label>
          <caption>
            <p>Patient characteristics.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="30"/>
            <col width="260"/>
            <col width="160"/>
            <col width="180"/>
            <col width="160"/>
            <col width="0"/>
            <col width="110"/>
            <col width="0"/>
            <col width="100"/>
            <thead>
              <tr valign="top">
                <td colspan="2">Characteristic</td>
                <td colspan="7">Grouped by 30-day all-cause mortality</td>
              </tr>
              <tr valign="top">
                <td colspan="2">
                  <break/>
                </td>
                <td>Overall</td>
                <td>Survived</td>
                <td>Deceased</td>
                <td colspan="2">Missing, n</td>
                <td colspan="2"><italic>P</italic> value</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td colspan="2">Patients, n (%)</td>
                <td>9496 (100)</td>
                <td>9124 (96.08)</td>
                <td>372 (3.92)</td>
                <td colspan="2">0</td>
                <td colspan="2">N/A<sup>a</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Age (years), mean (SD)</td>
                <td>42.2 (22.2)</td>
                <td>41.2 (21.8)</td>
                <td>66.8 (19.9)</td>
                <td colspan="2">0</td>
                <td colspan="2">&#60;.001<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="8">
                  <bold>Sex, n (%)</bold>
                </td>
                <td>.62<sup>c</sup></td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Female</td>
                <td>3188 (33.6)</td>
                <td>3068 (33.6)</td>
                <td>120 (32.3)</td>
                <td colspan="2">0</td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Male</td>
                <td>6308 (66.4)</td>
                <td>6056 (66.4)</td>
                <td>252 (67.7)</td>
                <td colspan="2">0</td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td colspan="2">Trajectory duration (days), mean (SD)</td>
                <td>7.6 (21.5)</td>
                <td>7.7 (21.9)</td>
                <td>5.4 (6.1)</td>
                <td colspan="2">0</td>
                <td colspan="2">&#60;.001<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Elixhauser Comorbidity Index, mean (SD)</td>
                <td>0.7 (3.2)</td>
                <td>0.6 (3.0)</td>
                <td>3.6 (5.8)</td>
                <td colspan="2">0</td>
                <td colspan="2">&#60;.001<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Height (cm), mean (SD)</td>
                <td>168.8 (25.0)</td>
                <td>168.6 (25.4)</td>
                <td>173.0 (13.0)</td>
                <td colspan="2">3895</td>
                <td colspan="2">&#60;.001<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Weight (kg), mean (SD)</td>
                <td>72.2 (23.7)</td>
                <td>71.9 (24.0)</td>
                <td>76.9 (18.5)</td>
                <td colspan="2">3223</td>
                <td colspan="2">&#60;.001<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Injury Severity Score, mean (SD)</td>
                <td>6.6 (10.6)</td>
                <td>6.0 (9.5)</td>
                <td>19.3 (20.8)</td>
                <td colspan="2">3941</td>
                <td colspan="2">&#60;.001<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Glasgow Coma Score, median (IQR)</td>
                <td>15.0 (13.0-15.0)</td>
                <td>15.0 (13.0-15.0)</td>
                <td>3.0 (3.0-6.0)</td>
                <td colspan="2">116</td>
                <td colspan="2">&#60;.001<sup>d</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">EWS<sup>e</sup> score, mean (SD)</td>
                <td>3.1 (2.8)</td>
                <td>3.0 (2.7)</td>
                <td>8.1 (3.8)</td>
                <td colspan="2">3840</td>
                <td colspan="2">&#60;.001<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Heart rate (aggregated mean; bpm), mean (SD)</td>
                <td>85.9 (16.3)</td>
                <td>85.8 (16.1)</td>
                <td>89.1 (20.6)</td>
                <td colspan="2">31</td>
                <td colspan="2">.003<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Respiratory rate (aggregated mean; per minute), median (IQR)</td>
                <td>16.4 (15.4-18.0)</td>
                <td>16.3 (15.4-18.0)</td>
                <td>16.7 (14.0-19.4)</td>
                <td colspan="2">1708</td>
                <td colspan="2">.84<sup>d</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Systolic blood pressure (aggregated mean; mmHg), mean (SD)</td>
                <td>126.1 (19.4)</td>
                <td>126.3 (19.2)</td>
                <td>123.1 (24.5)</td>
                <td colspan="2">47</td>
                <td colspan="2">.02<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Oxygen saturation (aggregated mean; %), median (IQR)</td>
                <td>97.5 (96.1-98.5)</td>
                <td>97.5 (96.2-98.6)</td>
                <td>95.5 (92.1-97.2)</td>
                <td colspan="2">32</td>
                <td colspan="2">&#60;.001<sup>d</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Temperature (aggregated mean; °C), median (IQR)</td>
                <td>36.9 (36.6-37.2)</td>
                <td>36.9 (36.6-37.2)</td>
                <td>37.0 (36.4-37.4)</td>
                <td colspan="2">2495</td>
                <td colspan="2">.23<sup>d</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Anesthetics (count of prescriptions), mean (minimum-maximum)</td>
                <td>3.9 (0.0-199.0)</td>
                <td>3.8 (0.0-199.0)</td>
                <td>6.4 (0.0-70.0)</td>
                <td colspan="2">0</td>
                <td colspan="2">&#60;.001<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Antibiotics (count of prescriptions), mean (minimum-maximum)</td>
                <td>12.8 (0.0-882.0)</td>
                <td>12.7 (0.0-882.0)</td>
                <td>15.0 (0.0-152.0)</td>
                <td colspan="2">0</td>
                <td colspan="2">.07<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Antithrombotic agents (count of prescriptions), mean (minimum-maximum)</td>
                <td>5.5 (0.0-601.0)</td>
                <td>5.6 (0.0-601.0)</td>
                <td>3.5 (0.0-41.0)</td>
                <td colspan="2">0</td>
                <td colspan="2">&#60;.001<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Blood products (count of prescriptions), mean (minimum-maximum)</td>
                <td>0.2 (0.0-27.0)</td>
                <td>0.2 (0.0-21.0)</td>
                <td>0.6 (0.0-27.0)</td>
                <td colspan="2">0</td>
                <td colspan="2">&#60;.001<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Cardiovascular drugs (count of prescriptions), mean (minimum-maximum)</td>
                <td>5.2 (0.0-839.0)</td>
                <td>5.1 (0.0-839.0)</td>
                <td>7.9 (0.0-175.0)</td>
                <td colspan="2">0</td>
                <td colspan="2">&#60;.001<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Diuretics (count of prescriptions), mean (minimum-maximum)</td>
                <td>2.2 (0.0-496.0)</td>
                <td>2.1 (0.0-496.0)</td>
                <td>5.1 (0.0-83.0)</td>
                <td colspan="2">0</td>
                <td colspan="2">&#60;.001<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Hemostatics (count of prescriptions), mean (minimum-maximum)</td>
                <td>0.4 (0.0-94.0)</td>
                <td>0.4 (0.0-94.0)</td>
                <td>1.1 (0.0-22.0)</td>
                <td colspan="2">0</td>
                <td colspan="2">&#60;.001<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Infusion (count of prescriptions), mean (minimum-maximum)</td>
                <td>4.4 (0.0-584.0)</td>
                <td>4.2 (0.0-584.0)</td>
                <td>7.9 (0.0-71.0)</td>
                <td colspan="2">0</td>
                <td colspan="2">&#60;.001<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Insulin (count of prescriptions), mean (minimum-maximum)</td>
                <td>1.5 (0.0-627.0)</td>
                <td>1.4 (0.0-627.0)</td>
                <td>3.5 (0.0-74.0)</td>
                <td colspan="2">0</td>
                <td colspan="2">&#60;.001<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Local anesthetics (count of prescriptions), mean (minimum-maximum)</td>
                <td>0.6 (0.0-57.0)</td>
                <td>0.6 (0.0-57.0)</td>
                <td>0.5 (0.0-20.0)</td>
                <td colspan="2">0</td>
                <td colspan="2">.05<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Neurological drugs (count of prescriptions), mean (minimum-maximum)</td>
                <td>10.9 (0.0-903.0)</td>
                <td>11.1 (0.0-903.0)</td>
                <td>6.1 (0.0-84.0)</td>
                <td colspan="2">0</td>
                <td colspan="2">&#60;.001<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Opioids (count of prescriptions), mean (minimum-maximum)</td>
                <td>13.0 (0.0-617.0)</td>
                <td>13.1 (0.0-617.0)</td>
                <td>9.4 (0.0-86.0)</td>
                <td colspan="2">0</td>
                <td colspan="2">&#60;.001<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Base excess (mmol/L), mean (SD)</td>
                <td>1.3 (4.0)</td>
                <td>1.4 (3.7)</td>
                <td>0.8 (6.8)</td>
                <td colspan="2">4074</td>
                <td colspan="2">.13<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Lactate (mmol/L), median (IQR)</td>
                <td>2.2 (1.6-3.2)</td>
                <td>2.2 (1.5-3.1)</td>
                <td>3.9 (2.3-8.4)</td>
                <td colspan="2">2269</td>
                <td colspan="2">&#60;.001<sup>d</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Hemoglobin (mmol/L), mean (SD)</td>
                <td>8.7 (1.0)</td>
                <td>8.7 (1.0)</td>
                <td>8.0 (1.4)</td>
                <td colspan="2">711</td>
                <td colspan="2">&#60;.001<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">Leukocyte count (10<sup>9</sup>/L), median (IQR)</td>
                <td>10.3 (7.7-14.2)</td>
                <td>10.1 (7.6-13.9)</td>
                <td>15.4 (11.1-20.6)</td>
                <td colspan="2">811</td>
                <td colspan="2">&#60;.001<sup>d</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">TEG<sup>f</sup> reaction time (minute), median (IQR)</td>
                <td>5.3 (4.7-6.3)</td>
                <td>5.3 (4.6-6.2)</td>
                <td>6.0 (4.8-7.5)</td>
                <td colspan="2">8279</td>
                <td colspan="2">&#60;.001<sup>d</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">TEG maximum amplitude (mm), mean (SD)</td>
                <td>65.4 (6.9)</td>
                <td>65.5 (6.7)</td>
                <td>64.6 (8.5)</td>
                <td colspan="2">8279</td>
                <td colspan="2">.30<sup>b</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="2">TEG fibrinolysis (LY30<sup>g</sup>; %), median (IQR)</td>
                <td>0.3 (0.0-1.0)</td>
                <td>0.3 (0.0-1.1)</td>
                <td>0.0 (0.0-0.6)</td>
                <td colspan="2">8393</td>
                <td colspan="2">&#60;.001<sup>d</sup></td>
              </tr>
              <tr valign="top">
                <td colspan="6">
                  <bold>A: Airway, n (%)</bold>
                </td>
                <td colspan="2">1165</td>
                <td>&#60;.001<sup>c</sup></td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Clear</td>
                <td>7970 (95.7)</td>
                <td>7786 (97.1)</td>
                <td>184 (58.8)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Threatened</td>
                <td>327 (3.9)</td>
                <td>226 (2.8)</td>
                <td>101 (32.3)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Blocked</td>
                <td>34 (0.4)</td>
                <td>6 (0.1)</td>
                <td>28 (8.9)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td colspan="6">
                  <bold>B: Breathing, n (%)</bold>
                </td>
                <td colspan="2">1151</td>
                <td>&#60;.001<sup>c</sup></td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Normal</td>
                <td>7012 (84.0)</td>
                <td>6891 (85.8)</td>
                <td>121 (38.7)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Lightly affected</td>
                <td>1029 (12.3)</td>
                <td>952 (11.9)</td>
                <td>77 (24.6)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Very affected</td>
                <td>236 (2.8)</td>
                <td>175 (2.2)</td>
                <td>61 (19.5)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Respiratory arrest (ceased)</td>
                <td>68 (0.8)</td>
                <td>14 (0.2)</td>
                <td>54 (17.3)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td colspan="6">
                  <bold>C: Circulation, n (%)</bold>
                </td>
                <td colspan="2">1158</td>
                <td>&#60;.001<sup>c</sup></td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Normal</td>
                <td>6996 (83.9)</td>
                <td>6874 (85.6)</td>
                <td>122 (39.1)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Lightly affected</td>
                <td>1018 (12.2)</td>
                <td>931 (11.6)</td>
                <td>87 (27.9)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Very affected</td>
                <td>255 (3.1)</td>
                <td>205 (2.6)</td>
                <td>50 (16.0)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Cardiac arrest</td>
                <td>69 (0.8)</td>
                <td>16 (0.2)</td>
                <td>53 (17.0)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td colspan="6">
                  <bold>D: Disability, n (%)</bold>
                </td>
                <td colspan="2">1172</td>
                <td>&#60;.001<sup>c</sup></td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Awake</td>
                <td>6703 (80.5)</td>
                <td>6613 (82.5)</td>
                <td>90 (29.0)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Reduced awareness</td>
                <td>1237 (14.9)</td>
                <td>1161 (14.5)</td>
                <td>76 (24.5)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Unconscious</td>
                <td>384 (4.6)</td>
                <td>240 (3.0)</td>
                <td>144 (46.5)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td colspan="6">
                  <bold>Level 1 trauma bay visit, n (%)</bold>
                </td>
                <td colspan="2">0</td>
                <td>&#60;.001<sup>c</sup></td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>No</td>
                <td>4866 (51.2)</td>
                <td>4774 (52.3)</td>
                <td>92 (24.7)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Yes</td>
                <td>4630 (48.8)</td>
                <td>4350 (47.7)</td>
                <td>280 (75.3)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td colspan="6">
                  <bold>Operating room visit, n (%)</bold>
                </td>
                <td colspan="2">0</td>
                <td>&#60;.001<sup>c</sup></td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>No</td>
                <td>7228 (76.1)</td>
                <td>6974 (76.4)</td>
                <td>254 (68.3)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Yes</td>
                <td>2268 (23.9)</td>
                <td>2150 (23.6)</td>
                <td>118 (31.7)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td colspan="6">
                  <bold>ICU admission, n (%)</bold>
                </td>
                <td colspan="2">0</td>
                <td>&#60;.001<sup>c</sup></td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>No</td>
                <td>7902 (83.2)</td>
                <td>7704 (84.4)</td>
                <td>198 (53.2)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Yes</td>
                <td>1594 (16.8)</td>
                <td>1420 (15.6)</td>
                <td>174 (46.8)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td colspan="6">
                  <bold>Ward admission, n (%)</bold>
                </td>
                <td colspan="2">0</td>
                <td>.70<sup>c</sup></td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>No</td>
                <td>4337 (45.7)</td>
                <td>4163 (45.6)</td>
                <td>174 (46.8)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Yes</td>
                <td>5159 (54.3)</td>
                <td>4961 (54.4)</td>
                <td>198 (53.2)</td>
                <td colspan="2">
                  <break/>
                </td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table1fn1">
              <p><sup>a</sup>N/A: not applicable.</p>
            </fn>
            <fn id="table1fn2">
              <p><sup>b</sup>Welch <italic>t</italic> test.</p>
            </fn>
            <fn id="table1fn3">
              <p><sup>c</sup>Chi-square test.</p>
            </fn>
            <fn id="table1fn4">
              <p><sup>d</sup>Kruskal-Wallis test.</p>
            </fn>
            <fn id="table1fn5">
              <p><sup>e</sup>EWS: Early Warning Score.</p>
            </fn>
            <fn id="table1fn6">
              <p><sup>f</sup>TEG: thrombelastography.</p>
            </fn>
            <fn id="table1fn7">
              <p><sup>g</sup>LY30: Lysis-index 30.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <p>The training process used 7667 (80%) trauma cases, and 1829 (20%) cases were held out for independent validation. The temporally split holdout cohort showed notable distributional differences, most prominently a reduction in level 1 trauma bay contact, shorter trajectory durations, and lower rates of operating room visits and ward admissions, compared to the development cohort, while outcome prevalence remained comparable (Table S5 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p>
      </sec>
      <sec>
        <title>Performance</title>
        <p>Predicting 30-day mortality for the holdout dataset using full-length sequences, the model achieved an AUROC of 0.962 (95% CI 0.930-0.994) and an AUPRC of 0.655 (95% CI 0.545-0.765). ROC and PR curves at selected trajectory time points are presented in Figure S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
        <p>The model demonstrated sensitivity ranging from approximately 0.49 to 0.96 within the first 12 hours, with 5 to 25 top percent predictions marked as positives. Sensitivities across the trajectory for top 5 to 25 percent high-risk patients are presented in Figure S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
        <p>The model performance varied depending on prediction time in the patient trajectory. For population-level evaluation, the first 12 hours AUROC was 0.905-0.936 and AUPRC 0.388-0.461, from 12 to 72 hours AUROC was 0.936-0.954 and AUPRC 0.461-0.557, from 72 hours to 7 days AUROC was 0.954-0.957 and AUPRC 0.557-0.601, from 7 days to 30 days AUROC was 0.957-0.962 and AUPRC 0.601-0.655. For active-cohort evaluation, the first 12 hours AUROC was 0.887-0.926 and AUPRC 0.375-0.452, from 12 to 72 hours AUROC was 0.874-0.919 and AUPRC 0.375-0.409, from 72 hours to 7 days AUROC was 0.919-0.921 and AUPRC 0.316-0.409, from 7 days to 30 days AUROC was 0.905-0.921 and AUPRC 0.316-0.350. Both evaluation strategies with CIs and active cohort composition are presented in <xref rid="figure3" ref-type="fig">Figure 3</xref> (presented with CIs in Table S6 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p>
        <fig id="figure3" position="float">
          <label>Figure 3</label>
          <caption>
            <p>Dynamic evaluation of model performance and active cohort composition, starting at first prehospital data point registration. (A) Area under the receiver operating characteristic curve (AUROC) and area under the precision-recall curve (AUPRC) over the first 72 hours for population-level (all patients: solid lines) and active-cohort (dashed lines) evaluation strategies, with 95% CIs. (B) AUROC and AUPRC over 30 days using the same evaluation strategies. (C) Number of active patients (left axis) and 30-day mortality prevalence among the active cohort (right axis, %) over the first 72 hours. The dotted line indicates the total holdout population (n=1829). (D) Active patients and prevalence over 30 days.</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e86202_fig3.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Calibration and Clinical Use</title>
        <p>The raw model outputs exhibited overconfidence across evaluation time points. Post hoc calibration was performed using isotonic regression and Platt scaling, each fitted per time point on the development dataset and applied to the holdout predictions. Isotonic regression yielded lower ECE and Brier scores than Platt scaling beyond 1 hour and was selected for subsequent analyses (Figure S5 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Following isotonic calibration, low-probability bins tracked the diagonal more closely across time points, though reliability estimates became unstable at later time points (7 days and 14 days) due to sparsely populated calibration bins in the shrinking active cohort. Post hoc isotonic calibration reduced Brier scores from 0.037 to 0.027 at 1 hour and from 0.062 to 0.049 at 14 days. Reliability diagrams with Brier scores at selected time points are presented in Figure S6 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
        <p>Decision curve analysis using isotonic-calibrated outputs demonstrated positive net benefit over the “treat all” and “treat none” strategies across a range of clinically relevant threshold probabilities at all evaluated time points (1 hour, 6 hours, 12 hours, 3 days, 7 days, and 14 days; <xref rid="figure4" ref-type="fig">Figure 4</xref>). Net benefit was highest at low threshold probabilities and decreased as the threshold increased. At 7 and 14 days, the net benefit curves exhibited greater variability, consistent with the reduced active cohort size at later time points.</p>
        <fig id="figure4" position="float">
          <label>Figure 4</label>
          <caption>
            <p>Decision curve analysis showing net benefit of the model at selected evaluation time points (1 hour, 6 hours, 12 hours, 3 days, 7 days, and 14 days) across a range of threshold probabilities. Model outputs were calibrated using isotonic regression fitted per time point on the development dataset and applied to the active holdout set. Dashed lines represent the “treat all” strategy at each time point. The solid black line at zero represents the “treat none” strategy.</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e86202_fig4.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>Predicted probability distributions stratified by outcome are presented in Figure S7 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Across all time points, the distributions for deceased patients were concentrated at higher predicted probabilities than for survivors.</p>
        <p><italic>F<sub>β</sub></italic> optimized thresholds were calculated from internal cross-validation with the final hyperparameters. Using <italic>β</italic> of 5 (t=0.040), the model achieved 84% sensitivity and 93% specificity (Figure S8 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), with a PPV of 30%. In absolute numbers, this means identifying 56 of 67 deaths while wrongly labeling 128 cases. Using the harmonic mean of precision and sensitivity (<italic>β</italic>=1, t=0.500), the model achieved 46% sensitivity and 99% specificity (Figure S9 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), with a PPV of 76%. In absolute numbers, this identifies 31 of 67 deaths with 10 false positives.</p>
      </sec>
      <sec>
        <title>Comparison With Trauma Scores</title>
        <p>The comparisons were conducted on holdout patients eligible for each respective score, restricted to those with the registrations required for trauma score calculation, resulting in varying cohort sizes across time points and scores. TRISS requires ISS, which in our implementation enters the model sequentially as diagnosis codes are registered, but was made available to TRISS at the earliest available registration regardless of evaluation time point to maximize the comparable cohort.</p>
        <p>Compared with RTS (n=1401), the model achieved a mean AUROC advantage of +0.297, reaching significance (FDR-corrected <italic>P</italic>&#60;.05) at 50 of 54 evaluated time points (<xref rid="figure5" ref-type="fig">Figure 5</xref>).</p>
        <fig id="figure5" position="float">
          <label>Figure 5</label>
          <caption>
            <p>Statistical comparison of area under the receiver operating characteristic curve (AUROC) between the hybrid neural network (HNN) and Revised Trauma Score (RTS) using paired DeLong tests with Benjamini-Hochberg false discovery rate (FDR) correction. (A) ΔAUROC (HNN minus RTS) over the first 72 hours with 95% CIs. Green shading indicates time points with FDR-corrected significance (<italic>P</italic>&#60;.05). (B) ΔAUROC over 30 days. (C) FDR-corrected <italic>P</italic> values displayed as –log₁₀(p_adj) over the first 72 hours. Green points indicate significance below the α=0.05 threshold (dashed red line); gray points indicate nonsignificant comparisons. (D) FDR-corrected <italic>P</italic> values over 30 days.</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e86202_fig5.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>Compared with TRISS (n=878), the model achieved a mean AUROC advantage of +0.169, with significance at 46 of 54 time points (<xref rid="figure6" ref-type="fig">Figure 6</xref>). For both comparisons, significance was lost at later time points (beyond approximately 15 days). AUROC and AUPRC across time for each comparison are presented in Figures S10 and S11 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
        <fig id="figure6" position="float">
          <label>Figure 6</label>
          <caption>
            <p>Statistical comparison of area under the receiver operating characteristic curve (AUROC) between the hybrid neural network (HNN) and Trauma and Injury Severity Score (TRISS) using paired DeLong tests with Benjamini-Hochberg false discovery rate (FDR) correction. (A) ΔAUROC (HNN minus TRISS) over the first 72 hours with 95% CIs. Green shading indicates time points with FDR-corrected significance (<italic>P</italic>&#60;.05). (B) ΔAUROC over 30 days. (C) FDR-corrected <italic>P</italic> values displayed as –log₁₀(p_adj) over the first 72 hours. Green points indicate significance below the α=0.05 threshold (dashed red line); gray points indicate nonsignificant comparisons. (D) FDR-corrected <italic>P</italic> values over 30 days.</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e86202_fig6.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Model Behavior</title>
        <p>SHAP analysis at the cohort level revealed that static patient features contributed most to model predictions. Age was consistently among the highest-contributing features across all evaluated time frames (mean &#124;SHAP&#124; 0.171 at 1 hour, increasing to 0.465 at 14 days; <xref rid="figure7" ref-type="fig">Figure 7</xref>). ECI was the second-highest static continuous contributor, surpassing age at intermediate time frames (0.530 vs 0.455 at 7 days) before declining at later time points. Among static categorical features, the prehospital disability (D) assessment was the most important (0.136 at 1 hour, 0.445 at 7 days), followed by breathing (B) assessment and first hospital of admission.</p>
        <fig id="figure7" position="float">
          <label>Figure 7</label>
          <caption>
            <p>Cohort-level Shapley additive explanations (SHAP) analysis of feature importance across evaluation time frames using active-cohort patients computed using GradientExplainer with a background dataset of 1000 training patients. (A) Sum absolute SHA<italic>P</italic> value per measured cell for continuous and categorical time series (TS) features over time. (B) Top 15 time series features ranked by mean absolute SHAP per measured cell across the full trajectory, showing both continuous and categorical channels. (C) Mean absolute SHA<italic>P</italic> values per measured cell for categorical time series features across time frames. (D) Mean absolute SHAP per measured cell for continuous time series channels across time frames. (E) Mean absolute SHA<italic>P</italic> values for static categorical features across time frames. (F) Mean absolute SHA<italic>P</italic> values for static continuous features across time frames.</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e86202_fig7.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>Among time-series features, per-measurement importance was shared between continuous channels and categorical (medication and ADT) channels (<xref rid="figure7" ref-type="fig">Figures 7</xref>A-7D). Among continuous time-series channels, GCS had the highest per-measurement importance at early time points (0.0484 at 1 hour), followed by mean SpO<sub>2</sub> (0.0240) and heart rate variability (SD 0.0138). GCS had continuously high importance throughout available time frames. Lactate demonstrated high importance at 6 to 24 hours with decline, thereafter, followed by leukocytes displaying a similar pattern. ISS derived from clinical notes showed increasing contribution as the time frame expanded, becoming the most important continuous channel by 14 days (0.0214). Per-measurement importance of continuous time-series features decreased over time as the trajectory progressed (<xref rid="figure7" ref-type="fig">Figure 7</xref>A). Among categorical time-series channels, several medication categories dominated, with opioids and infusion fluids having the highest per-measurement importance at 12 hours (0.0564 and 0.0535, respectively), comparable in magnitude to the top continuous channels at the same time point (GCS: 0.0290). Cardiovascular drugs peaked at 3 days (0.0463), while hemostatics showed sustained importance across the full trajectory, peaking at 7 days (0.0398). On a per-measurement basis, the highest-contributing medication categories were comparable in magnitude to the top continuous channels.</p>
        <p>The relative contribution of static features increased over time while per-measurement importance of temporal features decreased. Full cohort-level SHA<italic>P</italic> values across time frames are presented in Table S7 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p>
      </sec>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <sec>
        <title>Principal Findings</title>
        <p>This study focused on constructing trajectories for patients with trauma automatically from available EHRs for dynamic risk prediction of 30-day mortality. We found that using a hybrid neural network for predicting 30-day mortality achieved an AUROC of 0.962 and an AUPRC of 0.655 on a holdout dataset using full-length sequences. Given the 30-day mortality rate of 3.92% (372/9496), the AUPRC substantially exceeds the baseline prevalence, indicating strong discriminative performance despite marked class imbalance. Raw model outputs exhibited overconfidence, which post hoc isotonic calibration substantially reduced, improving alignment with observed outcomes particularly at lower predicted probabilities, which is the range most relevant to ruling out deterioration in clinically stable patients.</p>
        <p>The dynamic evaluation of performance demonstrated varying predictive performance over time, with cumulative discrimination improving as trajectories progress. To characterize this under different assumptions, we used 2 complementary evaluation strategies: population-level evaluation, retaining all patients at every time point, and active-cohort evaluation, restricted to patients with trajectories extending beyond each time point. The patient trajectory duration followed a positively skewed distribution with most patients having less than 7 days duration (<xref rid="figure3" ref-type="fig">Figure 3</xref>), resulting in decreasing active cohort sizes and widening CIs at later time points, presenting an important aspect for interpreting the presented results. Due to the temporal holdout split, the last active patient died near day 21, making evaluation beyond this point unreliable or impossible.</p>
      </sec>
      <sec>
        <title>Model Behavior</title>
        <p>SHAP analysis at the cohort level revealed that static patient features, particularly age and ECI, contributed most to model predictions across the trajectory. Among time-series features, GCS, ISS, SpO<sub>2</sub>, and heart rate variability were the most important continuous channels. Categorical time-series features were dominated by medication categories, with cardiovascular drugs, infusion fluids, and hemostatics ranking among the highest per-measurement contributors. The high importance of medication features is clinically plausible, as the initiation and intensity of pharmacological intervention directly reflects the treating clinician’s assessment of patient severity, though the observational design precludes distinguishing prognostic signal from treatment-outcome confounding.</p>
        <p>Early, high SHA<italic>P</italic> values for vitals and GCS align with these being the primary data available during the prehospital phase, where prehospital ABCD assessments provided complementary information, with the disability (D) assessment the most important static categorical feature spanning across all time frames.</p>
        <p>Temporal analysis of feature importance showed that the relative contribution of static features increased at later time points while per-measurement importance of individual temporal features generally decreased, though total categorical time-series importance remained elevated through intermediate time frames. These findings support the clinical plausibility of the model’s learned representations and provide transparency into which data streams drive risk estimation at different phases of the trauma trajectory.</p>
      </sec>
      <sec>
        <title>Clinical Use and Implications</title>
        <p>In most clinical applications, the model output would be presented to clinicians as a calibrated probability with reference levels rather than a binary classification, functioning as a continuously updated severity index analogous to existing trauma scores but with sequential updating capability presenting risk trend over time. This approach avoids the information loss inherent in dichotomizing a continuous risk estimate and allows clinicians to interpret the output in the context of their own decision framework. However, certain applications require binary classification, such as selecting cases for quality-of-care review or triggering automated deterioration alerts. For screening applications that prioritize sensitivity, we optimized the threshold using <italic>F<sub>β</sub></italic> (<italic>β</italic>=5) on the validation dataset, achieving 84% sensitivity and 93% specificity, identifying 56 of 67 deaths while generating 128 false positives. For settings where actionable precision is paramount, optimizing by <italic>F</italic><sub>1</sub> yielded 76% PPV, identifying 31 of 67 deaths with only 10 false positives. The acceptable trade-off between sensitivity and false-positive burden will vary by use case and time point, and threshold selection should be guided by institutional priorities.</p>
        <p>Deploying sequential risk prediction in clinical practice introduces risks that must be addressed alongside discriminative performance. A primary concern is alert fatigue: if the model generates frequent or nonactionable alerts, clinicians may habituate to the output and disregard clinically meaningful signals. Because the model generates a prediction at every evaluation time point, the total alert volume per patient substantially exceeds that of a single-admission score. The threshold analyses presented above were conducted on full-sequence predictions, where each patient contributes a single prediction based on their complete trajectory, and therefore represent an upper bound on achievable precision. While the <italic>F</italic><sub>1</sub>-optimized threshold yielded only 10 false positives across the holdout set under these conditions, per–time point prediction with incomplete trajectories, particularly in the early phase where model uncertainty is highest, would likely produce a higher operational false-positive burden. A parallel concern applies at the late-trajectory tail, where sparse event density limits the stability of calibrated probabilities and warrants additional caution in any alerting logic that depends on probability magnitude rather than on relative risk ranking. Characterizing the per–time point alert rate under sequential deployment is an important direction for future implementation research. Strategies for mitigating alert fatigue include restricting alert triggers to threshold crossings or sustained risk elevations rather than displaying continuous output, suppressing repeated alerts for the same patient within a defined time window, and tailoring alert thresholds to clinical context (eg, lower thresholds in the trauma bay or higher thresholds on the ward).</p>
        <p>The risk of overreliance, where clinicians defer excessively to model output at the expense of independent clinical judgment, is equally relevant. The model is intended to augment, not replace, clinical decision-making. Model outputs should be interpreted as one input among many for clinical decision-making, particularly in the early trajectory where limited data yields wider uncertainty. Institutional safeguards such as structured training on model limitations, transparent communication of prediction uncertainty, and periodic auditing of decision patterns following model deployment are essential to ensure that the tool supports rather than replaces clinical reasoning. Prospective evaluation of these integration safeguards will form a core component of future implementation studies.</p>
        <p>The sequential prediction framework supports several clinical applications across the trauma care continuum. First, continuously updated risk estimates can serve as an urgency signal in the trauma bay, where the initial clinical picture is often incomplete and rapidly evolving. A rising predicted mortality in the first hours may prompt earlier escalation to definitive surgical intervention or activation of additional resources. Second, the model output can function as a structured communication tool during transitions between treatment phases (eg, from the trauma bay to the operating room, or from the operating room to the ICU), providing receiving teams with an objective, data-driven summary of trajectory severity that complements verbal handover. Third, in the operative setting, real-time risk estimates may inform treatment strategy decisions, such as the choice between damage-control and definitive surgery. Fourth, in the inpatient setting, continued risk assessment may serve as a patient deterioration alert or as a tool for prioritizing patients during ward rounds, enabling clinicians to allocate attention to those with rising or persistently elevated risk trajectories.</p>
        <p>It is important to note that although the model incorporates prehospital data, it is not intended for prehospital decision-making in the current scope. Patients declared dead on scene were excluded from the study cohort, and the model requires data from the hospital trajectory to generate meaningful sequential predictions. The target population is therefore patients who survive to in-hospital trauma team activation, with prehospital data serving as early contextual input that enriches the initial prediction rather than as a stand-alone prehospital triage tool.</p>
      </sec>
      <sec>
        <title>Comparison With Prior Work</title>
        <p>In this study, we directly compared the model performance to the 2 trauma scores: RTS and TRISS. Because each score requires specific registrations for calculation, comparisons were restricted to eligible patients at each time point, resulting in varying cohort sizes across time points and scores.</p>
        <p>Several additional caveats apply. First, TRISS requires ISS, which in our model enters sequentially as registrations become available. To maximize the comparable cohort, we provided ISS to TRISS from the earliest registration regardless of time point, effectively giving TRISS an informational advantage. Second, neither trauma score was designed for longitudinal risk prediction: RTS was developed for field triage, and TRISS for benchmarking trauma care quality against the Major Trauma Outcome Study cohort. Despite these advantages afforded to the baseline scores, the hybrid neural network significantly outperformed both across the majority of evaluation time points, with a mean AUROC advantage of +0.297 over RTS and +0.169 over TRISS. After FDR correction, the advantage was significant at 50 of the eligible 54 time points for RTS and 46 of 54 for TRISS. Statistical significance was lost beyond approximately 15 days, where decreasing active-cohort sizes reduced statistical power.</p>
        <p>Our model’s performance compares with recent advancements in trauma mortality prediction. The Parkland Trauma Index of Mortality (PTIM) [<xref ref-type="bibr" rid="ref27">27</xref>] is an ML algorithm predicting 48-hour mortality during the first 72 hours of hospitalization for patients with polytrauma, reporting an AUROC of 0.94 for 516 cases. PTIM uses 23 variables, including vital signs, laboratory values, clinical scores, and demographics. One notable advantage of the PTIM the integration into the EHR system, allowing for real-time, hourly updates of mortality predictions. This feature enhances its clinical use, as demonstrated in a survey-based study where 77% of providers reported that the PTIM assisted in their treatment decisions [<xref ref-type="bibr" rid="ref28">28</xref>].</p>
        <p>Moreover, the Epic Deterioration Index (EDI) [<xref ref-type="bibr" rid="ref29">29</xref>] also demonstrated impressive results, with an AUROC of 0.98 for predicting in-hospital mortality within 24 hours of admission for 1325 level 1 trauma center patients. Both the PTIM study and the EDI validation study operate on shorter time frames compared to our study, which focuses on 30-day mortality prediction. Moreover, our study included a larger and more heterogeneous population, whereas PTIM only included patients with polytrauma, and the EDI validation included only level 1 trauma center patients. Particularly, these differences limit direct performance comparisons but demonstrate the feasibility of ML-based mortality prediction models in trauma care.</p>
        <p>Traditionally, risk estimation often required manual data entry into separate interfaces, with information gathered from the EHR and then entered into a distinct system for risk calculation, such as the Predictive Optimal Trees in Emergency Surgery Risk calculator [<xref ref-type="bibr" rid="ref30">30</xref>]. In our approach, we have concentrated on input features that can be automatically collected, ensuring that risk predictions can be generated with minimal manual effort and made readily available. While difficult, this strategy reduces the burden on health care providers and enables real-time risk assessment and reduces the potential for human error [<xref ref-type="bibr" rid="ref31">31</xref>].</p>
      </sec>
      <sec>
        <title>Study Limitations</title>
        <p>This study has several limitations that should be considered when interpreting the results.</p>
        <p>First, the population consists of patients in a single region in eastern Denmark, thus not fully representing the entire demographic spectrum of patients with trauma and may contain population bias. Moreover, the data might reflect local practices and guidelines, such as in the predefined categories for ABCD assessment. Despite these specific categories being implemented and used nationally, there may be regional differences in how clinicians are instructed to conduct subjective assessments. Within the study period itself, comparison of baseline characteristics between the development and holdout cohorts (Table S5 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) revealed notable distributional differences between cohorts, most prominently a considerable reduction in level 1 trauma bay contact, alongside shorter trajectory durations and lower rates of operating room visits, while outcome prevalence remained stable and core injury characteristics including ISS and ECI were comparable. No changes to regional triage protocols were implemented during the study period, and the cause of this shift is unknown. Because the SHAP analysis identifies facility-based features among the contributors to model predictions, this reduction represents a shift in one of the model’s informative inputs, not merely a background cohort characteristic. If the lower rate of level 1 trauma bay contact reflects a genuine change in case mix, the feature retains its prognostic relevance; if it instead reflects unmeasured changes in referral patterns among patients of comparable severity, the learned association between trauma bay contact and mortality risk may not generalize to settings where such patterns have shifted further. Disentangling these mechanisms is not possible from observational data alone. That the model maintained strong discrimination despite this distributional shift provides some evidence of robustness, but prospective validation with contemporaneous cohort definitions is needed to confirm generalizability under continued system-level changes.</p>
        <p>If the model is to be used in other countries or regions, retraining with local data is highly advisable [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref32">32</xref>], as trauma systems and underlying patient demography vary substantially even between comparable Western health care systems. Further studies should investigate whether similar modeling approaches yield comparable predictive performance when using different structures of the same information. This research should focus on varied data representations (eg, structured vs unstructured data), alternative categorization schemes, and alternative encoding methods.</p>
        <p>Second, the binary classification framework does not account for time-to-event dynamics and introduces evaluation challenges for patients whose trajectories end before a given evaluation time point. We address this through dual evaluation strategies (population-level and active-cohort), but both carry inherent trade-offs: informative censoring in the former and survivorship bias in the latter.</p>
        <p>Third, an important consideration in interpreting the sequential evaluation results is the model’s framing structure. The prediction target, 30-day all-cause mortality from first contact with health care providers, remains fixed across all evaluation time points, while the observation window expands as more clinical data becomes available. This means that the observed improvement in discriminative performance over time reflects 2 concurrent effects: the model receiving richer input data, and the effective prediction horizon shrinking as the elapsed time since trauma activation increases. At early time points, the model must predict an outcome up to 30 days in the future with limited information; at later time points, much of the outcome-relevant period has already elapsed. Disentangling these 2 contributions is not possible within this study design and represents a limitation of the fixed–end point evaluation framework. We adopted this framing to support future extension to multiple clinical outcomes that are not naturally suited to time-to-event modeling, such as massive transfusion, mechanical ventilation, and posttrauma complications, for which binary classification provides a more generalizable structure.</p>
        <p>Fourth, the exclusion of patients declared dead on scene creates a survivor cohort that does not capture the most extreme end of the severity spectrum. This exclusion is intentional, as the model requires sequential hospital data to generate predictions and is therefore applicable only to patients surviving to hospital admission. However, it limits the model’s ability to learn from the highest-acuity prehospital presentations and should be considered when interpreting early-trajectory predictions.</p>
        <p>Fifth, post hoc calibration using isotonic regression was fitted on the training and validation set and applied to the holdout set, which together comprise 372 outcome events. Because isotonic regression is nonparametric, it is susceptible to overfitting in sparse regions of the input distribution, which in our setting correspond to the late-trajectory time points where the active cohort shrinks and both event counts and prediction diversity decline. The reliability diagrams at later time points reflect this instability. In a clinical deployment, overfit calibration at the late tail could translate into misleading probability estimates in precisely the region where the model is already most uncertain, compounding the alarm fatigue and overtriage risks discussed above. Mitigation strategies include restricting calibrated probability presentation to time points with adequate event density, adopting parametric alternatives such as logistic calibration for the late-trajectory region, or periodic recalibration as additional patient trajectories accumulate. Future work with larger validation cohorts would strengthen confidence in calibration and clinical use assessments across the full prediction horizon.</p>
        <p>Sixth, the inclusion of therapeutic interventions (including blood transfusions, vasopressors, and anesthetics) as sequential input features introduces the potential for treatment-outcome confounding, commonly referred to as the treatment paradox. The model may learn to associate aggressive interventions with higher mortality risk, reflecting the severity of the underlying condition rather than a causal effect of treatment. We consider this an intentional design choice: in clinical practice, the initiation of aggressive interventions (eg, massive transfusion protocol activation) itself conveys prognostic information, and experienced trauma clinicians interpret such interventions as indicators of severity that shape subsequent management decisions. The model is designed to mirror this clinical reasoning pattern, integrating treatment signals as markers of evolving patient state. SHAP analysis showed that medication-related features, particularly cardiovascular drugs, infusion fluids, and hemostatics, were among the highest-contributing categorical time-series channels on a per-measurement basis, indicating that the model does rely on treatment signals for its predictions. While the clinical plausibility of these associations is high (patients receiving aggressive pharmacological intervention are more severely injured), the observational nature of the training data precludes causal inference, and users should be aware that the model captures associations between treatment patterns and outcomes, not treatment effects. Disentangling the prognostic value of treatment signals from treatment-outcome confounding would require causal inference methods or experimental designs that are beyond the scope of this study.</p>
        <p>Seventh, whether the transformer-based architecture offers a performance advantage over simpler model classes (such as gradient-boosted trees or recurrent neural networks) remains an open empirical question. The objective of this study was to demonstrate the feasibility of sequential modeling for trauma risk prediction, not to identify the optimal architecture for this task. The transformer was selected for its capacity to process variable-length multivariate sequences with attention-based weighting, but simpler architectures may achieve comparable discrimination with lower computational cost and greater interpretability. Systematic architecture comparison represents a valuable direction for future work, ideally conducted on a shared benchmark dataset to enable direct comparison across studies.</p>
        <p>Eighth, SHA<italic>P</italic> values were calculated on subsets for both background (n=1000) and analysis (n≤500) rather than full cohorts, which may influence importance estimates, particularly given the low outcome prevalence. This was done due to computational limits. Full-cohort SHAP analysis and systematic ablation studies would be needed to confirm the relative feature importance rankings and to determine whether any input modalities could be excluded without performance degradation.</p>
        <p>Ninth, the temporal positioning of ISS within the sequential input relies on diagnosis registration dates for <italic>ICD</italic>-derived scores and note last-edited time stamps for free-text extractions. Both are imperfect proxies for the point at which the information was clinically derivable. Last-edited time stamps can postdate this point when notes are amended after initial documentation. This was a deliberate conservative choice against data leakage, with the trade-off that the model may receive the information later than it was truly derivable.</p>
        <p>Tenth, this study does not include a subgroup analysis of predictive performance across sociodemographic strata. The descriptive statistics indicate significant differences in age and sex distribution between survivors and decedents, and trauma care delivery is known to differ across demographic groups in ways that can propagate into model behavior. Potential sources of inequity include uneven representation of older patients and women in the training cohort, differential documentation patterns across prehospital services, and outcome label heterogeneity driven by differences in goals-of-care decisions near end of life. Evaluating whether the model achieves comparable discrimination, calibration, and net benefit across age brackets and between sexes is a necessary step before clinical deployment. We have deferred this analysis to the prospective validation protocol under development, where subgroup definitions, sample size requirements, and the fairness criterion (eg, equalized odds or calibration parity) can be prespecified against a defined clinical-use framework rather than tested post hoc on the holdout set. Until that evaluation is completed, model outputs should be interpreted with awareness that subgroup-specific performance has not been established.</p>
        <p>Eleventh, the benchmark comparisons against RTS and TRISS were restricted to patients for whom the respective scores were computable at each evaluation time point. RTS requires initial GCS, systolic blood pressure, and respiratory rate, all of which are systematically missing in patients receiving early airway management (including prehospital intubation) and in those with unobtainable vital signs due to hemodynamic collapse. Such patients are, on average, the most severely injured in the cohort. Their exclusion biases the RTS-eligible subcohort toward lower acuity and likely causes the observed performance gap between the hybrid neural network and RTS to underestimate the true advantage among the most critical patients, where both methods would be evaluated on a cohort with more homogeneous baseline risk. TRISS is subject to the same eligibility-driven bias through its dependence on RTS components, compounded by additional ISS and injury-mechanism requirements. The comparative metrics reported here should therefore be interpreted as a lower bound on the model’s advantage over traditional scores in the full trauma cohort.</p>
      </sec>
      <sec>
        <title>Future Directions</title>
        <p>While our model demonstrates strong predictive performance, there are several avenues for potential improvement and future research. This study focused on 30-day mortality, where predicting additional outcomes such as massive transfusion, ICU admission with mechanical ventilation, major surgical interventions, posttrauma complications, readmissions, or functional status could provide a more comprehensive risk assessment. Another application of this modeling approach was demonstrated by our research group using a prehospital EHR to predict surgical needs stratified into specialties [<xref ref-type="bibr" rid="ref33">33</xref>]. Adapting the model to predict several outcomes relevant at different decision-making branching points could significantly impact patient care from the earliest stages of trauma management to hospital discharge.</p>
        <p>Incorporating additional data types could also enhance the model’s predictive power. However, careful consideration must be given to avoid data leakage, particularly when including diagnoses and procedures that might be directly related to the outcome. Future studies should explore integrating such information while maintaining the integrity of the prediction task. Another area for improvement lies in exploring different data representations and aggregations. For instance, medications and laboratory results could be grouped or aggregated in various ways to capture more nuanced patterns. These alternative representations might reveal additional insights and improve the model’s performance. Research into optimal grouping strategies for different data types could yield significant advancements in predictive accuracy.</p>
        <p>Future studies should further scrutinize subgroup model performance and behavior analysis to identify optimal configurations between predictive performance and clinical use. Next-phase evaluation would assess the effect of our model in the clinic by randomized controlled trials prescribing our model, using well-established quality indicators as outcomes, such as those used in the Trauma Quality Improvement Program by the American College of Surgeons [<xref ref-type="bibr" rid="ref34">34</xref>]. This would quantify clinical benefits while identifying implementation barriers and care gaps. Our study demonstrates the feasibility and potential of constructing patient trajectories from EHRs for dynamic risk prediction, showcased by the model’s strong performance in predicting 30-day mortality. While there are clear areas for future improvement and expansion, this approach advances toward more data-driven, personalized trauma care. As we continue to refine and expand these models, their integration into clinical practice could support trauma care from prehospital triage to long-term management. While these tools enhance clinical judgment rather than replace it, their integration could improve trauma outcomes through timely, insight-driven decision support.</p>
      </sec>
      <sec>
        <title>Conclusions</title>
        <p>This study demonstrates that a transformer-based sequential model trained on EHR data spanning prehospital and in-hospital care can predict 30-day all-cause mortality in patients with trauma with strong discrimination across the care trajectory. The model significantly outperformed established trauma risk scores (RTS and TRISS) at the majority of evaluation time points, with the advantage increasing as longitudinal data accumulated. SHAP-based interpretability analysis indicated that predictions were driven primarily by static patient characteristics, physiological measurements, and treatment patterns, with prehospital clinical assessments providing complementary early information. Post hoc isotonic calibration yielded well-calibrated risk estimates suitable for presentation as a continuous severity index, and decision curve analysis demonstrated positive net benefit over default strategies across a range of clinically relevant threshold probabilities at multiple time frames. These findings support the feasibility of sequential transformer-based modeling for dynamic trauma risk prediction. Future work will focus on prospective evaluation of clinical integration, targeting urgency awareness, interphase communication, operative decision support, and inpatient deterioration detection in the acute trauma setting.</p>
      </sec>
    </sec>
  </body>
  <back>
    <app-group>
      <supplementary-material id="app1">
        <label>Multimedia Appendix 1</label>
        <p>Additional information, tables, and figures.</p>
        <media xlink:href="jmir_v28i1e86202_app1.docx" xlink:title="DOCX File , 1547 KB"/>
      </supplementary-material>
    </app-group>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">ABCD</term>
          <def>
            <p>airway, breathing, circulation, disability</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">ADT</term>
          <def>
            <p>admission, discharge, and transfer</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb3">ATC</term>
          <def>
            <p>Anatomical Therapeutic Chemical</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb4">AUPRC</term>
          <def>
            <p>area under the precision-recall curve</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb5">AUROC</term>
          <def>
            <p>area under the receiver operating characteristic curve</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb6">ECE</term>
          <def>
            <p>expected calibration error</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb7">ECI</term>
          <def>
            <p>Elixhauser Comorbidity Index</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb8">EDI</term>
          <def>
            <p>Epic Deterioration Index</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb9">EHR</term>
          <def>
            <p>electronic health record</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb10">FDR</term>
          <def>
            <p>false discovery rate</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb11">GCS</term>
          <def>
            <p>Glasgow Coma Score</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb12">ICD</term>
          <def>
            <p>International Classification of Diseases</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb13">ICU</term>
          <def>
            <p>intensive care unit</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb14">ISS</term>
          <def>
            <p>Injury Severity Score</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb15">ML</term>
          <def>
            <p>machine learning</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb16">PPV</term>
          <def>
            <p>positive predictive value</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb17">PR</term>
          <def>
            <p>precision-recall</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb18">PTIM</term>
          <def>
            <p>Parkland Trauma Index of Mortality</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb19">ROC</term>
          <def>
            <p>receiver operating characteristic</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb20">RTS</term>
          <def>
            <p>Revised Trauma Score</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb21">SHAP</term>
          <def>
            <p>Shapley additive explanations</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb22">TRISS</term>
          <def>
            <p>Trauma and Injury Severity Score</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>Generative AI was not used for generating this manuscript. Grammarly (Superhuman Platform Inc), which uses generative AI, was used for spell-checking and sentence structuring during drafting.</p>
    </ack>
    <notes>
      <title>Data Availability</title>
      <p>Due to confidentiality issues, we are not at liberty to share the electronic health record (pre- and in-hospital) data used in this study. Access to Danish health data is granted through the Danish Patient Safety Authority. EHR data can then be obtained by reasonable request through the relevant region.</p>
      <p>Project source code is available at the following GitHub repository: https://github.com/a-millarch/dynamic-risk-prediction.</p>
    </notes>
    <notes>
      <title>Funding</title>
      <p>The study was supported by grants from the Danish Victims Foundation (grant #21-610-00127) and the Novo Nordisk Foundation (grant #NNF19OC0055183) to MS.</p>
    </notes>
    <fn-group>
      <fn fn-type="con">
        <p>Conceptualization: ASM, IC, HK, FF, SSR, MS</p>
        <p>Data curation: ASM</p>
        <p>Formal analysis: ASM</p>
        <p>Investigation: ASM</p>
        <p>Methodology: ASM, IC, HK, MS</p>
        <p>Software: ASM</p>
        <p>Visualization: ASM</p>
        <p>Resources: FF</p>
        <p>Funding acquisition: MS</p>
        <p>Project administration: MS</p>
        <p>Supervision: MS</p>
        <p>Writing—original draft: ASM</p>
        <p>Writing—review and editing: ASM, IC, HK, FF, SSR, MS</p>
      </fn>
      <fn fn-type="conflict">
        <p>Author MS has cofounded Aiomic, a company developing AI models for the health care sector. This work does not relate to any commercial activities and is for research only.
      The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Salim</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Stein</surname>
              <given-names>DM</given-names>
            </name>
            <name name-style="western">
              <surname>Zarzaur</surname>
              <given-names>BL</given-names>
            </name>
            <name name-style="western">
              <surname>Livingston</surname>
              <given-names>DH</given-names>
            </name>
          </person-group>
          <article-title>Measuring long-term outcomes after injury: current issues and future directions</article-title>
          <source>Trauma Surg Acute Care Open</source>
          <year>2023</year>
          <volume>8</volume>
          <issue>1</issue>
          <fpage>e001068</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/36919026"/>
          </comment>
          <pub-id pub-id-type="doi">10.1136/tsaco-2022-001068</pub-id>
          <pub-id pub-id-type="medline">36919026</pub-id>
          <pub-id pub-id-type="pii">tsaco-2022-001068</pub-id>
          <pub-id pub-id-type="pmcid">PMC10008475</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lapp</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Roper</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Kavanagh</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Bouamrane</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Schraag</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Dynamic prediction of patient outcomes in the intensive care unit: a scoping review of the state-of-the-art</article-title>
          <source>J Intensive Care Med</source>
          <year>2023</year>
          <volume>38</volume>
          <issue>7</issue>
          <fpage>575</fpage>
          <lpage>591</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://journals.sagepub.com/doi/10.1177/08850666231166349?url_ver=Z39.88-2003&#38;rfr_id=ori:rid:crossref.org&#38;rfr_dat=cr_pub  0pubmed"/>
          </comment>
          <pub-id pub-id-type="doi">10.1177/08850666231166349</pub-id>
          <pub-id pub-id-type="medline">37016893</pub-id>
          <pub-id pub-id-type="pmcid">PMC10302367</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Preti</surname>
              <given-names>LM</given-names>
            </name>
            <name name-style="western">
              <surname>Ardito</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Compagni</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Petracca</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Cappellaro</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>Implementation of machine learning applications in health care organizations: systematic review of empirical studies</article-title>
          <source>J Med Internet Res</source>
          <year>2024</year>
          <volume>26</volume>
          <fpage>e55897</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2024//e55897/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/55897</pub-id>
          <pub-id pub-id-type="medline">39586084</pub-id>
          <pub-id pub-id-type="pii">v26i1e55897</pub-id>
          <pub-id pub-id-type="pmcid">PMC11629039</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Boyd</surname>
              <given-names>CR</given-names>
            </name>
            <name name-style="western">
              <surname>Tolson</surname>
              <given-names>MA</given-names>
            </name>
            <name name-style="western">
              <surname>Copes</surname>
              <given-names>WS</given-names>
            </name>
          </person-group>
          <article-title>Evaluating trauma care: the TRISS method. Trauma Score and the Injury Severity Score</article-title>
          <source>J Trauma</source>
          <year>1987</year>
          <volume>27</volume>
          <issue>4</issue>
          <fpage>370</fpage>
          <lpage>378</lpage>
          <pub-id pub-id-type="medline">3106646</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Champion</surname>
              <given-names>HR</given-names>
            </name>
            <name name-style="western">
              <surname>Sacco</surname>
              <given-names>WJ</given-names>
            </name>
            <name name-style="western">
              <surname>Copes</surname>
              <given-names>WS</given-names>
            </name>
            <name name-style="western">
              <surname>Gann</surname>
              <given-names>DS</given-names>
            </name>
            <name name-style="western">
              <surname>Gennarelli</surname>
              <given-names>TA</given-names>
            </name>
            <name name-style="western">
              <surname>Flanagan</surname>
              <given-names>ME</given-names>
            </name>
          </person-group>
          <article-title>A revision of the Trauma Score</article-title>
          <source>J Trauma</source>
          <year>1989</year>
          <volume>29</volume>
          <issue>5</issue>
          <fpage>623</fpage>
          <lpage>629</lpage>
          <pub-id pub-id-type="doi">10.1097/00005373-198905000-00017</pub-id>
          <pub-id pub-id-type="medline">2657085</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Baker</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>O’Neill</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Haddon</surname>
              <given-names>WJ</given-names>
            </name>
            <name name-style="western">
              <surname>Long</surname>
              <given-names>WB</given-names>
            </name>
          </person-group>
          <article-title>The injury severity score: a method for describing patients with multiple injuries and evaluating emergency care</article-title>
          <source>J Trauma</source>
          <year>1974</year>
          <volume>14</volume>
          <issue>3</issue>
          <fpage>187</fpage>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Teasdale</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Jennett</surname>
              <given-names>B</given-names>
            </name>
          </person-group>
          <article-title>Assessment of coma and impaired consciousness: a practical scale</article-title>
          <source>Lancet</source>
          <year>1974</year>
          <volume>2</volume>
          <issue>7872</issue>
          <fpage>81</fpage>
          <lpage>84</lpage>
          <pub-id pub-id-type="doi">10.1016/s0140-6736(74)91639-0</pub-id>
          <pub-id pub-id-type="medline">4136544</pub-id>
          <pub-id pub-id-type="pii">S0140-6736(74)91639-0</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Maurer</surname>
              <given-names>LR</given-names>
            </name>
            <name name-style="western">
              <surname>Bertsimas</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Bouardi</surname>
              <given-names>HT</given-names>
            </name>
            <name name-style="western">
              <surname>El Hechi</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>El Moheb</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Giannoutsou</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Zhuo</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Dunn</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Velmahos</surname>
              <given-names>GC</given-names>
            </name>
            <name name-style="western">
              <surname>Kaafarani</surname>
              <given-names>HMA</given-names>
            </name>
          </person-group>
          <article-title>Trauma outcome predictor: an artificial intelligence interactive smartphone tool to predict outcomes in trauma patients</article-title>
          <source>J Trauma Acute Care Surg</source>
          <year>2021</year>
          <volume>91</volume>
          <issue>1</issue>
          <fpage>93</fpage>
          <lpage>99</lpage>
          <pub-id pub-id-type="doi">10.1097/TA.0000000000003158</pub-id>
          <pub-id pub-id-type="medline">33755641</pub-id>
          <pub-id pub-id-type="pii">01586154-202107000-00015</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="web">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Millarch</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Bonde</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Bonde</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Klein</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Folke</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Rudolph</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Sillesen</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Assessing optimal methods for transferring machine learning models to low-volume and imbalanced clinical datasets experiences from predicting outcomes of Danish trauma patients</article-title>
          <source>Front Digit Health</source>
          <year>2024</year>
          <access-date>2024-02-21</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.frontiersin.org/journals/digital-health/articles/10.3389/fdgth.2023.1249258">https://www.frontiersin.org/journals/digital-health/articles/10.3389/fdgth.2023.1249258</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bonde</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Bonde</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Troelsen</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Sillesen</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Assessing the utility of a sliding-windows deep neural network approach for risk prediction of trauma patients</article-title>
          <source>Sci Rep</source>
          <year>2023</year>
          <volume>13</volume>
          <issue>1</issue>
          <fpage>5176</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41598-023-32453-3"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41598-023-32453-3</pub-id>
          <pub-id pub-id-type="medline">36997598</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41598-023-32453-3</pub-id>
          <pub-id pub-id-type="pmcid">PMC10063587</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Vaswani</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Shazeer</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Parmar</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Uszkoreit</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Jones</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Gomez</surname>
              <given-names>AN</given-names>
            </name>
            <name name-style="western">
              <surname>Kaiser</surname>
              <given-names>Ł</given-names>
            </name>
            <name name-style="western">
              <surname>Polosukhin</surname>
              <given-names>I</given-names>
            </name>
          </person-group>
          <article-title>Attention is all you need</article-title>
          <year>2017</year>
          <month>12</month>
          <day>04</day>
          <conf-name>31st Conference on Neural Information Processing Systems (NIPS 2017)</conf-name>
          <conf-date>2017 Dec 04</conf-date>
          <conf-loc>Long Beach, CA, USA</conf-loc>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lauritsen</surname>
              <given-names>SM</given-names>
            </name>
            <name name-style="western">
              <surname>Thiesson</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Jørgensen</surname>
              <given-names>MJ</given-names>
            </name>
            <name name-style="western">
              <surname>Riis</surname>
              <given-names>AH</given-names>
            </name>
            <name name-style="western">
              <surname>Espelund</surname>
              <given-names>US</given-names>
            </name>
            <name name-style="western">
              <surname>Weile</surname>
              <given-names>JB</given-names>
            </name>
            <name name-style="western">
              <surname>Lange</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>The framing of machine learning risk prediction models illustrated by evaluation of sepsis in general wards</article-title>
          <source>NPJ Digit Med</source>
          <year>2021</year>
          <volume>4</volume>
          <issue>1</issue>
          <fpage>158</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41746-021-00529-x"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-021-00529-x</pub-id>
          <pub-id pub-id-type="medline">34782696</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-021-00529-x</pub-id>
          <pub-id pub-id-type="pmcid">PMC8593052</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Elixhauser</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Steiner</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Harris</surname>
              <given-names>DR</given-names>
            </name>
            <name name-style="western">
              <surname>Coffey</surname>
              <given-names>RM</given-names>
            </name>
          </person-group>
          <article-title>Comorbidity measures for use with administrative data</article-title>
          <source>Med Care</source>
          <year>1998</year>
          <volume>36</volume>
          <issue>1</issue>
          <fpage>8</fpage>
          <lpage>27</lpage>
          <pub-id pub-id-type="doi">10.1097/00005650-199801000-00004</pub-id>
          <pub-id pub-id-type="medline">9431328</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gasparini</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Comorbidity: an R package for computing comorbidity scores</article-title>
          <source>J Open Source Softw</source>
          <year>2018</year>
          <volume>3</volume>
          <issue>23</issue>
          <fpage>648</fpage>
          <pub-id pub-id-type="doi">10.21105/joss.00648</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="web">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Black</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Clark</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <source>icdpicr: “ICD” programs for injury categorization in R</source>
          <access-date>2024-12-16</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://cran.r-project.org/web/packages/icdpicr/index.html">https://cran.r-project.org/web/packages/icdpicr/index.html</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Eskesen</surname>
              <given-names>TO</given-names>
            </name>
            <name name-style="western">
              <surname>Sillesen</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Rasmussen</surname>
              <given-names>LS</given-names>
            </name>
            <name name-style="western">
              <surname>Steinmetz</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Agreement between standard and ICD-10-based injury severity scores</article-title>
          <source>Clin Epidemiol</source>
          <year>2022</year>
          <volume>14</volume>
          <fpage>201</fpage>
          <lpage>210</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/35221725"/>
          </comment>
          <pub-id pub-id-type="doi">10.2147/CLEP.S344302</pub-id>
          <pub-id pub-id-type="medline">35221725</pub-id>
          <pub-id pub-id-type="pii">344302</pub-id>
          <pub-id pub-id-type="pmcid">PMC8864409</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Thim</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Krarup</surname>
              <given-names>NHV</given-names>
            </name>
            <name name-style="western">
              <surname>Grove</surname>
              <given-names>EL</given-names>
            </name>
            <name name-style="western">
              <surname>Rohde</surname>
              <given-names>CV</given-names>
            </name>
            <name name-style="western">
              <surname>Løfgren</surname>
              <given-names>Bo</given-names>
            </name>
          </person-group>
          <article-title>Initial assessment and treatment with the airway, breathing, circulation, disability, exposure (ABCDE) approach</article-title>
          <source>Int J Gen Med</source>
          <year>2012</year>
          <volume>5</volume>
          <fpage>117</fpage>
          <lpage>121</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.tandfonline.com/doi/10.2147/IJGM.S28478?url_ver=Z39.88-2003&#38;rfr_id=ori:rid:crossref.org&#38;rfr_dat=cr_pub  0pubmed"/>
          </comment>
          <pub-id pub-id-type="doi">10.2147/IJGM.S28478</pub-id>
          <pub-id pub-id-type="medline">22319249</pub-id>
          <pub-id pub-id-type="pii">ijgm-5-117</pub-id>
          <pub-id pub-id-type="pmcid">PMC3273374</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref18">
        <label>18</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Stone</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Cross-validatory choice and assessment of statistical predictions</article-title>
          <source>J R Stat Soc Series B Stat Methodol</source>
          <year>1974</year>
          <volume>36</volume>
          <issue>2</issue>
          <fpage>147</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.google.com/goto?url=CAEScAHrOzAVVrGzrWrZPvvs5Qm64devFtGlIBD7_SZpkzxWZmjKPZY213Kdxs3o-KaYHN378Q8WvPabw3S0ntQtL-M3h1QQP0PwRLkJhHeiuuEcL50Xea3sicmFrwccpyFHljiYmHJsRhXLo68_Kd7mZBk"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref19">
        <label>19</label>
        <nlm-citation citation-type="web">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Oguiza</surname>
              <given-names>I</given-names>
            </name>
          </person-group>
          <source>Tsai - A State-of-the-Art Deep Learning Library for Time Series and Sequential Data</source>
          <year>2023</year>
          <access-date>2026-08-21</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://github.com/timeseriesAI/tsai">https://github.com/timeseriesAI/tsai</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref20">
        <label>20</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Khetan</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Cvitkovic</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Karnin</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <article-title>TabTransformer: tabular data modeling using contextual embeddings</article-title>
          <source>arXiv. Preprint posted online on December 11, 2020</source>
          <pub-id pub-id-type="doi">10.48550/arXiv.2012.06678</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref21">
        <label>21</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Calster</surname>
              <given-names>BV</given-names>
            </name>
            <name name-style="western">
              <surname>Collins</surname>
              <given-names>GS</given-names>
            </name>
            <name name-style="western">
              <surname>Vickers</surname>
              <given-names>AJ</given-names>
            </name>
            <name name-style="western">
              <surname>Wynants</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Kerr</surname>
              <given-names>KF</given-names>
            </name>
            <name name-style="western">
              <surname>Barreñada</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Varoquaux</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Singh</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Moons</surname>
              <given-names>KG</given-names>
            </name>
            <name name-style="western">
              <surname>Hernandez-Boussard</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Timmerman</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>McLernon</surname>
              <given-names>DJ</given-names>
            </name>
            <name name-style="western">
              <surname>van Smeden</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Steyerberg</surname>
              <given-names>EW</given-names>
            </name>
          </person-group>
          <article-title>Evaluation of performance measures in predictive artificial intelligence models to support medical decisions: overview and guidance</article-title>
          <source>Lancet Digit Health</source>
          <year>2025</year>
          <volume>7</volume>
          <issue>12</issue>
          <fpage>100916</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S2589-7500(25)00098-6"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.landig.2025.100916</pub-id>
          <pub-id pub-id-type="medline">41391983</pub-id>
          <pub-id pub-id-type="pii">S2589-7500(25)00098-6</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref22">
        <label>22</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Brier</surname>
              <given-names>GW</given-names>
            </name>
          </person-group>
          <article-title>Verification of forecasts expressed in terms of probability</article-title>
          <source>Mon Weather Rev</source>
          <year>1950</year>
          <volume>78</volume>
          <issue>1</issue>
          <fpage>1</fpage>
          <lpage>3</lpage>
          <pub-id pub-id-type="doi">10.1175/1520-0493(1950)078%3C0001:VOFEIT%3E2.0.CO;2</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref23">
        <label>23</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>DeLong</surname>
              <given-names>ER</given-names>
            </name>
            <name name-style="western">
              <surname>DeLong</surname>
              <given-names>DM</given-names>
            </name>
            <name name-style="western">
              <surname>Clarke-Pearson</surname>
              <given-names>DL</given-names>
            </name>
          </person-group>
          <article-title>Comparing the areas under two or more correlated receiver operating characteristic curves: a nonparametric approach</article-title>
          <source>Biometrics</source>
          <year>1988</year>
          <volume>44</volume>
          <issue>3</issue>
          <fpage>837</fpage>
          <lpage>845</lpage>
          <pub-id pub-id-type="medline">3203132</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref24">
        <label>24</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Benjamini</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Hochberg</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>Controlling the false discovery rate: a practical and powerful approach to multiple testing</article-title>
          <source>J R Stat Soc Series B Stat Methodol</source>
          <year>2018</year>
          <volume>57</volume>
          <issue>1</issue>
          <fpage>289</fpage>
          <lpage>300</lpage>
          <pub-id pub-id-type="doi">10.1111/j.2517-6161.1995.tb02031.x</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref25">
        <label>25</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lundberg</surname>
              <given-names>SM</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>SI</given-names>
            </name>
          </person-group>
          <article-title>A unified approach to interpreting model predictions</article-title>
          <source>arXiv. Preprint posted online on May 22, 2017</source>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.48550/arXiv.1705.07874"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref26">
        <label>26</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Erion</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Janizek</surname>
              <given-names>JD</given-names>
            </name>
            <name name-style="western">
              <surname>Sturmfels</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Lundberg</surname>
              <given-names>SM</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>SI</given-names>
            </name>
          </person-group>
          <article-title>Improving performance of deep learning models with axiomatic attribution priors and expected gradients</article-title>
          <source>Nat Mach Intell</source>
          <year>2021</year>
          <volume>3</volume>
          <issue>7</issue>
          <fpage>620</fpage>
          <lpage>631</lpage>
          <pub-id pub-id-type="doi">10.1038/s42256-021-00343-w</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref27">
        <label>27</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Starr</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Julka</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Nethi</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Watkins</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Fairchild</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Rinehart</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Park</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Dumas</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Box</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Cripps</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Parkland trauma index of mortality: real-time predictive model for trauma patients</article-title>
          <source>J Orthop Trauma</source>
          <year>2022</year>
          <volume>36</volume>
          <issue>6</issue>
          <fpage>280</fpage>
          <pub-id pub-id-type="doi">10.1097/bot.0000000000002290</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref28">
        <label>28</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Park</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Loza-Avalos</surname>
              <given-names>SE</given-names>
            </name>
            <name name-style="western">
              <surname>Harvey</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Hirschkorn</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Dultz</surname>
              <given-names>LA</given-names>
            </name>
            <name name-style="western">
              <surname>Dumas</surname>
              <given-names>RP</given-names>
            </name>
            <name name-style="western">
              <surname>Sanders</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Chowdhry</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Starr</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Cripps</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>A real-time automated machine learning algorithm for predicting mortality in trauma patients: survey says it's ready for prime-time</article-title>
          <source>Am Surg</source>
          <year>2024</year>
          <volume>90</volume>
          <issue>4</issue>
          <fpage>655</fpage>
          <lpage>661</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://journals.sagepub.com/doi/10.1177/00031348231207299?url_ver=Z39.88-2003&#38;rfr_id=ori:rid:crossref.org&#38;rfr_dat=cr_pub  0pubmed"/>
          </comment>
          <pub-id pub-id-type="doi">10.1177/00031348231207299</pub-id>
          <pub-id pub-id-type="medline">37848176</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref29">
        <label>29</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Mou</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Godat</surname>
              <given-names>LN</given-names>
            </name>
            <name name-style="western">
              <surname>El-Kareh</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Berndtson</surname>
              <given-names>AE</given-names>
            </name>
            <name name-style="western">
              <surname>Doucet</surname>
              <given-names>JJ</given-names>
            </name>
            <name name-style="western">
              <surname>Costantini</surname>
              <given-names>TW</given-names>
            </name>
          </person-group>
          <article-title>Electronic health record machine learning model predicts trauma inpatient mortality in real time: a validation study</article-title>
          <source>J Trauma Acute Care Surg</source>
          <year>2022</year>
          <volume>92</volume>
          <issue>1</issue>
          <fpage>74</fpage>
          <lpage>80</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/34932043"/>
          </comment>
          <pub-id pub-id-type="doi">10.1097/TA.0000000000003431</pub-id>
          <pub-id pub-id-type="medline">34932043</pub-id>
          <pub-id pub-id-type="pii">01586154-202201000-00013</pub-id>
          <pub-id pub-id-type="pmcid">PMC9032917</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref30">
        <label>30</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bertsimas</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Dunn</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Velmahos</surname>
              <given-names>GC</given-names>
            </name>
            <name name-style="western">
              <surname>Kaafarani</surname>
              <given-names>HMA</given-names>
            </name>
          </person-group>
          <article-title>Surgical risk is not linear: derivation and validation of a novel, user-friendly, and machine-learning-based predictive OpTimal Trees in Emergency Surgery Risk (POTTER) calculator</article-title>
          <source>Ann Surg</source>
          <year>2018</year>
          <volume>268</volume>
          <issue>4</issue>
          <fpage>574</fpage>
          <lpage>583</lpage>
          <pub-id pub-id-type="doi">10.1097/SLA.0000000000002956</pub-id>
          <pub-id pub-id-type="medline">30124479</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref31">
        <label>31</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Knevel</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Liao</surname>
              <given-names>KP</given-names>
            </name>
          </person-group>
          <article-title>From real-world electronic health record data to real-world results using artificial intelligence</article-title>
          <source>Ann Rheum Dis</source>
          <year>2023</year>
          <volume>82</volume>
          <issue>3</issue>
          <fpage>306</fpage>
          <lpage>311</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S0003-4967(24)08501-7"/>
          </comment>
          <pub-id pub-id-type="doi">10.1136/ard-2022-222626</pub-id>
          <pub-id pub-id-type="medline">36150748</pub-id>
          <pub-id pub-id-type="pii">S0003-4967(24)08501-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC9933153</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref32">
        <label>32</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Soltan</surname>
              <given-names>AAS</given-names>
            </name>
            <name name-style="western">
              <surname>Clifton</surname>
              <given-names>DA</given-names>
            </name>
          </person-group>
          <article-title>Machine learning generalizability across healthcare settings: insights from multi-site COVID-19 screening</article-title>
          <source>NPJ Digit Med</source>
          <year>2022</year>
          <volume>5</volume>
          <issue>1</issue>
          <fpage>69</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41746-022-00614-9"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-022-00614-9</pub-id>
          <pub-id pub-id-type="medline">35672368</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-022-00614-9</pub-id>
          <pub-id pub-id-type="pmcid">PMC9174159</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref33">
        <label>33</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Millarch</surname>
              <given-names>AS</given-names>
            </name>
            <name name-style="western">
              <surname>Folke</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Rudolph</surname>
              <given-names>SS</given-names>
            </name>
            <name name-style="western">
              <surname>Kaafarani</surname>
              <given-names>HM</given-names>
            </name>
            <name name-style="western">
              <surname>Sillesen</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Prehospital triage of trauma patients: predicting major surgery using artificial intelligence as decision support</article-title>
          <source>Br J Surg</source>
          <year>2025</year>
          <volume>112</volume>
          <issue>4</issue>
          <fpage>znaf058</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://academic.oup.com/bjs/article-lookup/doi/10.1093/bjs/znaf058"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/bjs/znaf058</pub-id>
          <pub-id pub-id-type="medline">40200724</pub-id>
          <pub-id pub-id-type="pii">8108908</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref34">
        <label>34</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Nathens</surname>
              <given-names>AB</given-names>
            </name>
            <name name-style="western">
              <surname>Cryer</surname>
              <given-names>HG</given-names>
            </name>
            <name name-style="western">
              <surname>Fildes</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>The American College of Surgeons trauma quality improvement program</article-title>
          <source>Surg Clin North Am</source>
          <year>2012</year>
          <volume>92</volume>
          <issue>2</issue>
          <fpage>441</fpage>
          <lpage>454</lpage>
          <pub-id pub-id-type="doi">10.1016/j.suc.2012.01.003</pub-id>
          <pub-id pub-id-type="medline">22414421</pub-id>
          <pub-id pub-id-type="pii">S0039-6109(12)00015-1</pub-id>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
