<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e94837</article-id><article-id pub-id-type="doi">10.2196/94837</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Development and Validation of an Interpretable Machine Learning Model for Staging <italic>Helicobacter pylori</italic>&#x2013;Initiated Intestinal-Type Gastric Cancer in the Correa Cascade: Cross-Sectional Study</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Tang</surname><given-names>Jiawei</given-names></name><degrees>MMed</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Chen</surname><given-names>Huijin</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Wenwen</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Tay</surname><given-names>Alfred Chin Yen</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Marshall</surname><given-names>Barry J</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ma</surname><given-names>Cong</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Wang</surname><given-names>Liang</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref><xref ref-type="aff" rid="aff7">7</xref><xref ref-type="aff" rid="aff8">8</xref></contrib></contrib-group><aff id="aff1"><institution>The Marshall Centre for Interventions in Infectious Disease, The University of Western Australia</institution><addr-line>Perth</addr-line><country>Australia</country></aff><aff id="aff2"><institution>Department of Laboratory Medicine, Shengli Oilfield Central Hospital</institution><addr-line>Dongyin</addr-line><addr-line>Shandong</addr-line><country>China</country></aff><aff id="aff3"><institution>Department of Clinical Medicine, School of the 1st Clinical Medicine, Xuzhou Medical University</institution><addr-line>Xuzhou</addr-line><addr-line>Jiangsu</addr-line><country>China</country></aff><aff id="aff4"><institution>Marshall Research Centre for Medical Microbial Biotechnology, Department of Life Sciences, Faculty of Science, The Hong Kong Polytechnic University (PolyU)</institution><addr-line>Hongkong</addr-line><country>China (Hong Kong)</country></aff><aff id="aff5"><institution>Department of Laboratory Medicine, Guangdong Provincial People's Hospital, Guangdong Academy of Medical Sciences, Southern Medical University</institution><addr-line>No. 106, Zhongshan 2nd Road, Yuexiu District</addr-line><addr-line>Guangzhou</addr-line><addr-line>Guangdong</addr-line><country>China</country></aff><aff id="aff6"><institution>School of Medicine, South China University of Technology</institution><addr-line>Guangzhou</addr-line><addr-line>Guangdong</addr-line><country>China</country></aff><aff id="aff7"><institution>Department of Intelligent Medical Engineering, School of Medical Informatics and Engineering, Xuzhou Medical University</institution><addr-line>Xuzhou</addr-line><addr-line>Jiangsu</addr-line><country>China</country></aff><aff id="aff8"><institution>School of Molecular Sciences, School of Biomedical Sciences, The University of Western Australia</institution><addr-line>Perth</addr-line><country>Australia</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Shingru</surname><given-names>Pratik</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Li</surname><given-names>Zhen</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Liang Wang, PhD, Department of Laboratory Medicine, Guangdong Provincial People's Hospital, Guangdong Academy of Medical Sciences, Southern Medical University, No. 106, Zhongshan 2nd Road, Yuexiu District, Guangzhou, Guangdong, 510080, China, 86 13921750542; <email>healthscience@foxmail.com</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>7</day><month>8</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e94837</elocation-id><history><date date-type="received"><day>07</day><month>03</month><year>2026</year></date><date date-type="rev-recd"><day>14</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>14</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Jiawei Tang, Huijin Chen, Wenwen Zhang, Alfred Chin Yen Tay, Barry J Marshall, Cong Ma, Liang Wang. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 7.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e94837"/><abstract><sec><title>Background</title><p>Gastric cancer (GC) is one of the most common malignant tumors worldwide, with <italic>Helicobacter pylori</italic>&#x2013;associated intestinal-type gastric cancer (IGC) being the most prevalent subtype, accounting for approximately 85% of cases. Because most patients are diagnosed at intermediate or advanced stages, early screening and accurate stage stratification of IGC progression remain major clinical challenges.</p></sec><sec><title>Objective</title><p>This study aimed to develop an interpretable machine learning (ML) model that leverages routine laboratory indicators to perform stage-specific diagnosis for patients across different stages of IGC.</p></sec><sec sec-type="methods"><title>Methods</title><p>Data from 2180 patients with known <italic>H pylori</italic> infection status were collected at 2 centers and included healthy controls (HCs), nonatrophic gastritis, atrophic gastritis, intestinal metaplasia, and GC. After excluding cases with severe (&#x003E;25%) missing data, 1784 patients were included for model development and validation. Data imputation and feature selection were performed, and synthetic minority oversampling technique (SMOTE) augmentation was applied to the internal training dataset to improve the diagnostic performance of the model. Six ML algorithms were developed. Model performance and clinical decision-making utility were evaluated using multiple metrics and approaches, while Shapley Additive Explanations (SHAP)&#x2013;based interpretability was used to identify key indicators and provide threshold reference values for them. Finally, a web-based tool was developed based on the Streamlit platform.</p></sec><sec sec-type="results"><title>Results</title><p>Through feature selection, 27 features were ultimately retained for final model construction. Among the 6 algorithms, CatBoost (categorical boosting) demonstrated the best performance, achieving an internal validation accuracy of 80.91%, sensitivity of 78.57%, and specificity of 95.27%. In the Guangdong Provincial People&#x2019;s Hospital (GDPH) and Shengli Oilfield Central Hospital (SOCH) external validation cohorts, CatBoost maintained robust performance, with accuracies of 79.96% and 83.37%, sensitivities of 76.82% and 84.91%, specificities of 93.14% and 95.73%, and area under the curves (AUCs) of 0.94 and 0.97, respectively. Confusion matrix analysis showed that the model was particularly reliable in identifying extreme disease states, including HCs and GC, whereas misclassifications mainly occurred between adjacent intermediate pathological stages. Calibration curves and Brier scores indicated good agreement between predicted and observed outcomes. Decision curve analysis (DCA) further confirmed the clinical net benefit across relevant threshold ranges. SHAP-based interpretability analysis identified monocyte count (MONO%), albumin/globulin ratio (A/G), basophil percentage (BASO%), platelet distribution width (PDW), total bilirubin (DBIL), neutrophil count (NEUT#), age, lymphocyte count (LYMPH#), creatinine (CREA), and aspartate aminotransferase (AST) as important contributors, reflecting inflammatory, hematological, nutritional, and metabolic changes during IGC progression. Based on these features, a lightweight predictive model was developed and deployed as a web-based application to facilitate translational and practical applications.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>This study developed an interpretable ML model based on routine laboratory data for stage-specific prediction of <italic>H pylori</italic>&#x2013;associated IGC progression, with promising applicability as an auxiliary diagnostic tool.</p></sec></abstract><kwd-group><kwd>intestinal-type gastric cancer</kwd><kwd>routine laboratory indicators</kwd><kwd>machine learning</kwd><kwd>hematological parameters</kwd><kwd>Correa cascade</kwd><kwd>Helicobacter pylori</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>According to the latest GLOBOCAN estimates from the International Agency for Research on Cancer (IARC), approximately 968,000 new gastric cancer (GC) cases and 660,000 related deaths occurred worldwide in 2022. Both the incidence and mortality of GC rank fifth globally [<xref ref-type="bibr" rid="ref1">1</xref>]. East Asia, Eastern Europe, and South America are high-incidence regions for GC, with the disease burden particularly pronounced in East Asian countries. Data from the National Cancer Center of China (NCC) indicate that approximately 358,700 new GC cases and about 260,400 deaths occurred in China in 2022 [<xref ref-type="bibr" rid="ref2">2</xref>]. From a histopathological perspective, Danish pathologist Pekka Antero Laur&#x00E9;n classified GC into intestinal-type gastric cancer (IGC) and diffuse-type GC in 1965 [<xref ref-type="bibr" rid="ref3">3</xref>]. <italic>Helicobacter pylori</italic> infection is considered the initiating and essential factor for the development of IGC, as it colonizes the gastric mucosa, triggering active gastritis [<xref ref-type="bibr" rid="ref4">4</xref>], and gradually driving disease progression to atrophic gastritis (AG), intestinal metaplasia (IM), dysplasia, and ultimately IGC [<xref ref-type="bibr" rid="ref5">5</xref>]. Due to the insidious clinical symptoms of early GC, most patients are diagnosed at an advanced stage, missing the optimal opportunity for curative surgery. Even among those who undergo surgical treatment, approximately two-thirds remain at risk of recurrence or metastasis. Large-scale cohort studies suggest that the optimal window for GC intervention lies before the onset of IM [<xref ref-type="bibr" rid="ref6">6</xref>]. Therefore, early detection and intervention of IGC is crucial for improving patient prognosis [<xref ref-type="bibr" rid="ref7">7</xref>].</p><p>Endoscopic examination is the cornerstone of GC diagnosis, but conventional white-light gastroscopy heavily depends on the operator&#x2019;s experience, making it difficult to promptly identify gastric mucosal atrophy and IM [<xref ref-type="bibr" rid="ref8">8</xref>]. Although a combined biopsy is the gold standard for GC diagnosis [<xref ref-type="bibr" rid="ref9">9</xref>], its invasiveness and the expensive medical resources limit its use in widespread screening. Multiomics analyses based on the microbiome and metabolome can characterize differential microbial communities and metabolites during disease progression. These alterations may provide potential biomarkers for the early diagnosis of IGC [<xref ref-type="bibr" rid="ref10">10</xref>]. However, similar to endoscopy, these methods are costly and time-consuming, making them unsuitable for large-scale, on-site screening. AI-based pathological image analysis can assist clinicians in diagnosis and has been proven capable of accurately detecting GC [<xref ref-type="bibr" rid="ref11">11</xref>]. For example, Song et al [<xref ref-type="bibr" rid="ref12">12</xref>] trained a deep learning model using 2123 pixel-level stained whole-slide images, achieving nearly 100% sensitivity and 80.6% specificity in real-world test data. However, high-quality AI diagnostic models for pathology rely on massive manual annotations, requiring substantial labor, time, and computing power, which hinders the iterative development of such technologies. Hematologic indices, owing to their clinical accessibility and quantifiability, have increasingly been used to explore noninvasive biomarkers of inflammation and cancer. For example, red cell distribution width (RDW) has been associated with atherosclerosis, inflammatory bowel disease, and multiple cancers [<xref ref-type="bibr" rid="ref13">13</xref>]. Mean corpuscular volume (MCV) has been used to assess the prognosis of esophageal cancer [<xref ref-type="bibr" rid="ref14">14</xref>], whereas hematocrit (HCT) performs comparably to hemoglobin (Hb) in predicting mortality risk in triple-negative breast cancer [<xref ref-type="bibr" rid="ref15">15</xref>]. Because immune-inflammatory responses can promote angiogenesis and stimulate tumor cell proliferation, invasion, and metastasis, they are directly related to tumor initiation and progression [<xref ref-type="bibr" rid="ref16">16</xref>]. Current hematologic studies on the early diagnosis of GC have mainly focused on the diagnostic and prognostic value of inflammatory markers [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>]. Studies show that peripheral blood inflammatory indices are associated with overall survival in patients with locally advanced GC. Platelet-to-lymphocyte ratio (PLR), neutrophil-to-lymphocyte ratio (NLR), and lymphocyte counts often indicate a poorer prognosis [<xref ref-type="bibr" rid="ref19">19</xref>]. Meanwhile, inflammatory markers such as interleukin-6 (IL-6), C-reactive protein (CRP), and procalcitonin (PCT) are markedly elevated in GC and can provide auxiliary evidence for diagnosis and staging [<xref ref-type="bibr" rid="ref20">20</xref>]. In addition, several routine blood count parameters have been reported: RDW is increased, while PDW is decreased in patients with GC [<xref ref-type="bibr" rid="ref13">13</xref>]. A long-term follow-up study has also found that an elevated white blood cell (WBC) count is associated with an increased risk of GC, particularly among individuals infected with <italic>H pylori</italic> [<xref ref-type="bibr" rid="ref21">21</xref>]. However, most existing studies relied on conventional statistical methods and analyzed only a limited number of hematologic indices, making it difficult to capture complex interindicator associations and latent patterns. Therefore, more advanced computational approaches are urgently needed to fully realize the potential of hematologic indices in the early identification and stratified diagnosis of GC.</p><p>Unlike simple statistical analyses based on a single or a few indicators, machine learning (ML) techniques, with their ability to efficiently model multidimensional features and perform pattern recognition, can be deeply integrated with hematological parameters, greatly expanding their clinical diagnostic value [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. For example, Yang et al [<xref ref-type="bibr" rid="ref1">1</xref>] analyzed 57 hematological indicators from patients with nasopharyngeal, esophageal, and lung cancers undergoing immunosuppressive therapy, and the ML model achieved a sensitivity of 85% and a specificity of 79%; Zhang et al [<xref ref-type="bibr" rid="ref24">24</xref>] integrated 58 hematological and biochemical indicators from 2951 samples to construct an ML model, obtaining a sensitivity of 85.44% and a specificity of 83.82% in the cross-validation cohort. However, the feasibility of these indicators for staging-specific diagnosis of IGC within the framework of the Correa cascade has not yet been investigated. This retrospective multicenter study aimed to develop and validate an interpretable ML-based diagnostic framework. It was designed for stage-specific diagnosis of <italic>H pylori</italic>&#x2013;associated IGC progression along the Correa cascade using routine laboratory indicators. Specifically, we used demographic information, complete blood count parameters, and various biochemical indicators from 2 independent centers. We sought to compare imputation methods and data augmentation techniques in terms of data quality, optimize and evaluate multiple ML algorithms, identify the most suitable diagnostic model, and interpret the contribution of key clinical indicators to model decision-making. Subsequently, we aimed to develop a user-friendly online diagnostic platform for the preliminary screening of IGC at different stages of the Correa cascade. This model is expected to assist in the early screening of different stages of IGC progression, especially in resource-limited primary health care settings.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Population</title><p>This retrospective study aimed to construct and validate a predictive model for classifying disease stages along the <italic>H pylori</italic>&#x2013;associated Correa cascade. Data were obtained from the laboratory information systems (LIS) of 2 independent, large-scale, grade 3A hospitals, Guangdong Provincial People&#x2019;s Hospital (GDPH) and Shengli Oilfield Central Hospital (SOCH), between January 2022 and May 2025. The collected data included demographic information, complete blood counts, and serum biochemical markers. <italic>H pylori</italic> infection status was confirmed through gastric mucosal histological examination. Participants were classified into 5 groups, including healthy control (HC), nonatrophic gastritis (NAG), AG, IM, and GC. All diagnoses were independently determined by 2 experienced pathologists according to pathological standards. HCs were recruited from individuals undergoing routine health examinations who had normal upper gastrointestinal endoscopic findings and negative <italic>H pylori</italic> test results. The same exclusion criteria were applied to both HCs and patients across all disease stages, including a history of gastric resection or other gastrointestinal surgeries, coexisting malignancies, a history of organ transplantation, or severe heart, lung, liver, kidney, or hematologic diseases.</p></sec><sec id="s2-2"><title>Data Preprocessing</title><p>Features with more than 25% missing values were excluded. Missing values in the remaining features were imputed using 6 methods, including mean imputation, median imputation, k-nearest neighbor (KNN) imputation, linear interpolation, multiple imputation, and regression imputation. The imputation results were compared using box plots and histograms to evaluate the impact of different imputation methods on the data distribution and statistical characteristics. Furthermore, considering the potential influence of multicollinearity between features on prediction accuracy, features with a high correlation in the Spearman correlation analysis were excluded if they had a lower correlation with the target variable. Ultimately, the following features were retained, including gender, age, aspartate aminotransferase (AST), albumin (ALB), albumin/globulin ratio (A/G), total bilirubin (DBIL), alkaline phosphatase (AKP), gamma-glutamyl transferase (GGT), glucose (GLU), urea (UREA), creatinine (CREA), uric acid (UA), total cholesterol (TC), triglycerides (TG), high-density lipoprotein cholesterol (HDL), monocyte percentage (MONO%), basophil percentage (BASO%), neutrophil count (NEUT#), lymphocyte count (LYMPH#), monocyte count (MONO#), eosinophil count (EO#), red blood cell (RBC) count, mean corpuscular hemoglobin (MCH), mean corpuscular hemoglobin concentration (MCHC), platelet count (PLT), mean platelet volume (MPV), and platelet distribution width (PDW). To address the issue of class imbalance, we used the synthetic minority over-sampling technique (SMOTE) to generate synthetic samples. SMOTE was applied only to the internal training set, whereas the internal validation set and the 2 external validation cohorts were kept unaugmented. By reducing the bias toward the majority class during model training, this approach enables more stable learning of minority-class patterns and effectively improves the model&#x2019;s sensitivity to rare but clinically meaningful cases [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>]. In subsequent analyses, we compared and evaluated the effectiveness of this method.</p></sec><sec id="s2-3"><title>Model Selection and Performance Evaluation</title><p>This study was conducted using Python 3.9.7, along with the <italic>scikit-learn</italic>, <italic>catboost</italic>, <italic>lightgbm</italic>, and <italic>xgboost</italic> packages for training ML models. The 6 ML algorithms used included adaptive boosting (AdaBoost, <italic>AdaBoostClassifier</italic>), categorical boosting (CatBoost, <italic>CatBoostClassifier</italic>), decision tree (DT, <italic>DecisionTreeClassifier</italic>), light gradient boosting machine (LGBM, <italic>LGBMClassifier</italic>), random forest (RF, <italic>RandomForestClassifier</italic>), and eXtreme gradient boosting (XGBoost, <italic>XGBClassifier</italic>). The training cohort from 2 hospitals was divided into 2 subsets, with 80% allocated as the internal training set and 20% as the internal validation set. Before formal training, all models underwent hyperparameter tuning using <italic>GridSearchCV</italic> to fit all combinations of model parameters (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), followed by 5-fold cross-validation (CV) for optimal hyperparameter selection. Each model was then trained using the optimal parameter combination. Performance was evaluated via accuracy, macro precision, specificity, macro recall (sensitivity), and macro <italic>F</italic><sub>1</sub>-score. Receiver operating characteristic (ROC) curve and confusion matrix were also generated to evaluate discrimination and class-specific performance. All models followed the same construction and optimization process to assess the quality of the new data generated by SMOTE preprocessing, thus validating the feasibility of this approach. The best diagnostic model, selected after comparison of all models, was further evaluated using a calibration curve to determine whether the model&#x2019;s predicted probabilities align with the actual outcomes. Additionally, the model&#x2019;s potential clinical value was measured using decision curve analysis (DCA). Finally, an external validation cohort was simultaneously collected from both hospitals, with inclusion and exclusion criteria that matched those of the training cohort. The external validation data were entered into the best diagnostic model, and the same evaluation metrics were used to assess the model&#x2019;s performance on unseen data.</p></sec><sec id="s2-4"><title>Interpretability Analysis</title><p>To effectively interpret the model&#x2019;s decision results, the study uses Shapley Additive Explanations (SHAP) for interpretability analysis. SHAP values for each feature are calculated using <italic>TreeExplainer</italic> from the SHAP library. Feature importance is ranked using the <italic>summary_plot</italic>, which provides a visual representation of the overall contribution of each feature to the model&#x2019;s predictions. The <italic>dependence_plot</italic> function further analyzes the impact of individual features on model predictions, examining the trend of feature changes across different categories and their relationship with the SHAP values. The ridge plot displays the distribution of each feature across different groups, visually highlighting the differences between categories.</p></sec><sec id="s2-5"><title>Web Page Deployment Tool Based on Streamlit Framework</title><p>To facilitate the clinical application of the developed method, we integrate the best model, constructed from the top 10 features identified by SHAP, into a tool developed using the <italic>Streamlit</italic> framework. A web page deployment tool is created for predictive analysis at different stages of the Correa model. Users can input the corresponding feature values into the model and click the &#x201C;Predict&#x201D; button. The tool will return the prediction result and save it as the most recent prediction record.</p></sec><sec id="s2-6"><title>Ethical Considerations</title><p>This study was approved by the ethics review committee of Guangdong Provincial People&#x2019;s Hospital (KY2025-1060-01). As this was a retrospective study, the requirement for informed consent was waived. To ensure privacy protection, all data were deidentified using participant codes, and no identifiable information was disclosed.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Patient Characteristics</title><p>This retrospective analysis included 2180 patients at different stages of <italic>H pylori</italic>&#x2013;associated GC along the Correa cascade, who were enrolled in the cohort for predictive model construction. During the study, GDPH and SOCH excluded 200 and 196 patients with extensive missing data, respectively. Ultimately, data from 1784 patients were used for model training and external validation. The training cohort included 1644 patients, consisting of 53 HC, 723 NAG, 181 AG, 606 IM, and 81 GC (<xref ref-type="table" rid="table1">Table 1</xref>). Baseline information after SMOTE augmentation is shown in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. At GDPH and SOCH, data from 50 and 90 patients were collected, respectively, as external validation cohorts. The overall framework of the study is illustrated in <xref ref-type="fig" rid="figure1">Figure 1</xref>, and the detailed study design workflow is provided in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>The overall framework of the study. The study primarily includes cohort recruitment, data preprocessing, model construction, and evaluation, as well as external validation and tool development. Ada: adaptive boosting; Cat: categorical boosting; DT: decision tree; GDPH: Guangdong Provincial People&#x2019;s Hospital; LGBM: light gradient boosting machine; RF: random forest; SOCH: Shengli Oilfield Central Hospital; XGB: eXtreme gradient boosting.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e94837_fig01.png"/></fig><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Baseline characteristics of training sets in Guangdong Provincial People&#x2019;s Hospital (GDPH) and Shengli Oilfield Central Hospital (SOCH) cohorts.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Characteristics</td><td align="left" valign="bottom" colspan="5">Training set (N=1644)</td><td align="left" valign="bottom"><italic>P</italic> value<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">HC (n=53)</td><td align="left" valign="bottom">NAG (n=723)</td><td align="left" valign="bottom">AG (n=181)</td><td align="left" valign="bottom">IM (n=606)</td><td align="left" valign="bottom">GC (n=81)</td><td align="left" valign="bottom"/></tr></thead><tbody><tr><td align="left" valign="top">Male, n (%)</td><td align="left" valign="top">19 (35.80)</td><td align="left" valign="top">431 (59.61)</td><td align="left" valign="top">104 (57.46)</td><td align="left" valign="top">404 (66.67)</td><td align="left" valign="top">51 (62.90)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Age (y)</td><td align="left" valign="top">57.66 (12.23)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">53.00 (47.00&#x2010;63.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">59.28 (11.19)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">57.50 (52.00&#x2010;68.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">62.54 (10.66)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">AST<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td><td align="left" valign="top">21.08 (4.40)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">20.00 (17.00&#x2010;24.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">21.00 (17.00&#x2010;25.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">21.00 (17.00&#x2010;25.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">19.00 (16.00&#x2010;23.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">.13</td></tr><tr><td align="left" valign="top">ALB<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup></td><td align="left" valign="top">42.45 (4.26)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">41.88 (39.20&#x2010;44.70)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">40.69 (38.15&#x2010;43.54)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">41.90 (39.50&#x2010;44.30)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">39.56 (5.33)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">A/G<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup></td><td align="left" valign="top">1.60 (1.50&#x2010;1.75)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">1.60 (1.40&#x2010;1.80)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">1.50 (1.30&#x2010;1.70)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">1.60 (1.50&#x2010;1.80)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">1.45 (0.28)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">DBIL<sup><xref ref-type="table-fn" rid="table1fn7">g</xref></sup></td><td align="left" valign="top">3.10 (2.35&#x2010;4.25)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">2.50 (1.90&#x2010;3.30)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">2.40 (1.88&#x2010;3.20)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">2.50 (2.00&#x2010;3.50)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">2.40 (1.70&#x2010;3.50)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">.003</td></tr><tr><td align="left" valign="top">AKP<sup><xref ref-type="table-fn" rid="table1fn8">h</xref></sup></td><td align="left" valign="top">74.00 (60.50&#x2010;87.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">69.00 (56.00&#x2010;84.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">71.00 (59.00&#x2010;88.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">70.00 (59.00&#x2010;84.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">72.00 (59.00&#x2010;90.02)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">.23</td></tr><tr><td align="left" valign="top">GGT<sup><xref ref-type="table-fn" rid="table1fn9">i</xref></sup></td><td align="left" valign="top">21.00 (15.00&#x2010;28.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">23.00 (16.00&#x2010;39.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">24.00 (15.00&#x2010;39.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">22.00 (16.00&#x2010;38.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">19.00 (15.00&#x2010;31.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">.15</td></tr><tr><td align="left" valign="top">GLU<sup><xref ref-type="table-fn" rid="table1fn10">j</xref></sup></td><td align="left" valign="top">5.11 (4.61&#x2010;5.70)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">5.11 (4.61&#x2010;5.78)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">5.14 (4.62&#x2010;5.93)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">5.12 (4.62&#x2010;5.88)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">5.14 (4.71&#x2010;6.05)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">.73</td></tr><tr><td align="left" valign="top">UREA<sup><xref ref-type="table-fn" rid="table1fn11">k</xref></sup></td><td align="left" valign="top">5.30 (3.94&#x2010;6.15)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">4.96 (4.28&#x2010;5.83)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">5.22 (4.23&#x2010;6.35)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">5.22 (4.43&#x2010;6.30)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">5.69 (4.61&#x2010;6.60)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">.001</td></tr><tr><td align="left" valign="top">CREA<sup><xref ref-type="table-fn" rid="table1fn12">l</xref></sup></td><td align="left" valign="top">55.80 (48.80&#x2010;67.10)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">64.70 (54.30&#x2010;76.80)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">66.60 (56.71&#x2010;82.63)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">66.80 (55.60&#x2010;76.93)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">67.00 (55.73&#x2010;78.60)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">UA<sup><xref ref-type="table-fn" rid="table1fn13">m</xref></sup></td><td align="left" valign="top">309.70 (105.00)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">336.80 (266.10&#x2010;403.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">345.00 (288.10&#x2010;414.95)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">330.10 (270.98&#x2010;396.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">309.82 (94.87)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">.004</td></tr><tr><td align="left" valign="top">TC<sup><xref ref-type="table-fn" rid="table1fn14">n</xref></sup></td><td align="left" valign="top">4.87 (0.99)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">4.95 (4.16&#x2010;5.69)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">5.11 (4.27&#x2010;5.73)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">4.88 (4.21&#x2010;5.62)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">4.67 (1.14)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">.10</td></tr><tr><td align="left" valign="top">TG<sup><xref ref-type="table-fn" rid="table1fn15">o</xref></sup></td><td align="left" valign="top">1.22 (0.92&#x2010;1.54)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">1.33 (0.98&#x2010;1.96)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">1.36 (0.97&#x2010;2.09)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">1.36 (0.99&#x2010;1.95)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">1.26 (0.91&#x2010;1.76)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">.24</td></tr><tr><td align="left" valign="top">HDL<sup><xref ref-type="table-fn" rid="table1fn16">p</xref></sup></td><td align="left" valign="top">1.42 (1.12&#x2010;1.68)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">1.19 (1.00&#x2010;1.46)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">1.23 (1.04&#x2010;1.43)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">1.2 (1.02&#x2010;1.47)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">1.43 (0.48)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">.03</td></tr><tr><td align="left" valign="top">MONO%<sup><xref ref-type="table-fn" rid="table1fn17">q</xref></sup></td><td align="left" valign="top">6.80 (5.50&#x2010;8.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">6.80 (5.00&#x2010;8.10)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">6.30 (0.11&#x2010;7.80)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">6.70 (5.50&#x2010;8.20)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">0.13 (0.07&#x2010;7.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">BASO%<sup><xref ref-type="table-fn" rid="table1fn18">r</xref></sup></td><td align="left" valign="top">0.40 (0.20&#x2010;0.60)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">0.30 (0.10&#x2010;0.50)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">0.30 (0.01&#x2010;0.50)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">0.40 (0.20&#x2010;0.60)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">0.01 (0.005&#x2010;0.30)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">NEUT#<sup><xref ref-type="table-fn" rid="table1fn19">s</xref></sup></td><td align="left" valign="top">3.06 (2.72&#x2010;3.63)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">3.27 (2.63&#x2010;4.35)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">3.38 (2.6&#x2010;4.07)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">3.43 (2.67&#x2010;4.35)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">4.23 (3.13&#x2010;5.57)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">LYMPH#<sup><xref ref-type="table-fn" rid="table1fn20">t</xref></sup></td><td align="left" valign="top">1.67 (1.49&#x2010;2.31)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">1.83 (1.48&#x2010;2.24)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">1.87 (1.52&#x2010;2.25)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">1.91 (1.46&#x2010;2.35)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">1.6 (1.33&#x2010;2.01)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">.007</td></tr><tr><td align="left" valign="top">MONO#<sup><xref ref-type="table-fn" rid="table1fn21">u</xref></sup></td><td align="left" valign="top">0.36 (0.29&#x2010;0.47)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">0.42 (0.32&#x2010;0.52)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">0.44 (0.33&#x2010;0.56)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">0.42 (0.34&#x2010;0.55)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">0.48 (0.38&#x2010;0.64)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">EO#<sup><xref ref-type="table-fn" rid="table1fn22">v</xref></sup></td><td align="left" valign="top">0.1 (0.06&#x2010;0.14)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">0.12 (0.07&#x2010;0.20)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">0.14 (0.06&#x2010;0.24)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">0.11 (0.06&#x2010;0.20)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">0.10 (0.05&#x2010;0.19)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">.04</td></tr><tr><td align="left" valign="top">RBC<sup><xref ref-type="table-fn" rid="table1fn23">w</xref></sup></td><td align="left" valign="top">4.46 (0.63)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">4.56 (4.20&#x2010;4.91)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">4.51 (4.13&#x2010;4.87)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">4.59 (4.25&#x2010;4.90)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">4.23 (0.73)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">.001</td></tr><tr><td align="left" valign="top">MCH<sup><xref ref-type="table-fn" rid="table1fn24">x</xref></sup></td><td align="left" valign="top">30.10 (29.30&#x2010;31.50)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">30.10 (29.30&#x2010;31.40)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">30.60 (29.35&#x2010;31.90)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">30.70 (29.60&#x2010;31.70)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">30.50 (28.90&#x2010;31.60)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">MCHC<sup><xref ref-type="table-fn" rid="table1fn25">y</xref></sup></td><td align="left" valign="top">335.77 (12.17)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">335.00 (327.00&#x2010;343.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">337.00 (326.00&#x2010;344.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">337.00 (330.00&#x2010;344.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">334.00 (325.00&#x2010;342.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">.002</td></tr><tr><td align="left" valign="top">PLT<sup><xref ref-type="table-fn" rid="table1fn26">z</xref></sup></td><td align="left" valign="top">229.00 (192.00&#x2010;264.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">228.00 (195.00&#x2010;276.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">220.00 (183.50&#x2010;269.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">216.50 (182.00&#x2010;262.25)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">237.00 (199.50&#x2010;307.50)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">.001</td></tr><tr><td align="left" valign="top">MPV<sup><xref ref-type="table-fn" rid="table1fn27">aa</xref></sup></td><td align="left" valign="top">10.15 (1.09)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">10.10 (9.40&#x2010;10.60)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">10.01 (1.15)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">9.90 (9.20&#x2010;10.60)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">9.70 (0.99)<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">.03</td></tr><tr><td align="left" valign="top">PDW<sup><xref ref-type="table-fn" rid="table1fn28">ab</xref></sup></td><td align="left" valign="top">13.60 (11.30&#x2010;15.95)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">11.90 (10.50&#x2010;14.00)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">11.90 (10.40&#x2010;13.65)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">11.70 (10.40&#x2010;14.68)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">10.90 (9.50&#x2010;13.10)<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">&#x003C;.001</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup><italic>P</italic> value: healthy control (HC) vs nonatrophic gastritis (NAG), atrophic gastritis (AG) vs intestinal metaplasia (IM) vs gastric cancer (GC).</p></fn><fn id="table1fn2"><p><sup>b</sup>Mean (SD).</p></fn><fn id="table1fn3"><p><sup>c</sup>Median (IQR).</p></fn><fn id="table1fn4"><p><sup>d</sup>AST: aspartate aminotransferase.</p></fn><fn id="table1fn5"><p><sup>e</sup>ALB: albumin.</p></fn><fn id="table1fn6"><p><sup>f</sup>A/G: albumin-to-globulin ratio.</p></fn><fn id="table1fn7"><p><sup>g</sup>DBIL: direct bilirubin.</p></fn><fn id="table1fn8"><p><sup>h</sup>AKP: alkaline phosphatase.</p></fn><fn id="table1fn9"><p><sup>i</sup>GGT: gamma-glutamyl transferase.</p></fn><fn id="table1fn10"><p><sup>j</sup>GLU: glucose.</p></fn><fn id="table1fn11"><p><sup>k</sup>UREA: urea.</p></fn><fn id="table1fn12"><p><sup>l</sup>CREA: creatinine.</p></fn><fn id="table1fn13"><p><sup>m</sup>UA: uric acid.</p></fn><fn id="table1fn14"><p><sup>n</sup>TC: total cholesterol.</p></fn><fn id="table1fn15"><p><sup>o</sup>TG: triglycerides.</p></fn><fn id="table1fn16"><p><sup>p</sup>HDL: high-density lipoprotein.</p></fn><fn id="table1fn17"><p><sup>q</sup>MONO%: monocyte percentage.</p></fn><fn id="table1fn18"><p><sup>r</sup>BASO%: basophil percentage.</p></fn><fn id="table1fn19"><p><sup>s</sup>NEUT#: neutrophil count.</p></fn><fn id="table1fn20"><p><sup>t</sup>LYMPH#: lymphocyte count.</p></fn><fn id="table1fn21"><p><sup>u</sup>MONO#: monocyte count.</p></fn><fn id="table1fn22"><p><sup>v</sup>EO#: eosinophil count.</p></fn><fn id="table1fn23"><p><sup>w</sup>RBC: red blood cell.</p></fn><fn id="table1fn24"><p><sup>x</sup>MCH: mean corpuscular hemoglobin.</p></fn><fn id="table1fn25"><p><sup>y</sup>MCHC: mean corpuscular hemoglobin concentration.</p></fn><fn id="table1fn26"><p><sup>z</sup>PLT: platelet count.</p></fn><fn id="table1fn27"><p><sup>aa</sup>MPV: mean platelet volume.</p></fn><fn id="table1fn28"><p><sup>ab</sup>PDW: platelet distribution width.</p></fn></table-wrap-foot></table-wrap><p>This study included data from patients with IGC at different stages, collected at GDPH and SOCH. The data were divided into 5 groups: HC, NAG, AG, IM, and GC. The baseline demographic and clinical characteristics of the training cohort are shown in <xref ref-type="table" rid="table1">Table 1</xref>. Significant differences in gender distribution and age were observed among the 5 groups, with both variables showing statistical significance (<italic>P</italic>&#x003C;.001). In the disease groups, the proportion of males was higher than in the HC group, particularly in the IM and GC groups. Age progressively increased with disease severity; the highest age values were observed in the GC group and relatively lower values in the NAG group. In contrast, indicators related to metabolism and liver function (AST, AKP, GGT, GLU, TC, and TG) did not show significant differences among the groups (<italic>P</italic>&#x003E;.05). Several biochemical and hematological parameters exhibited significant differences across disease stages and demonstrated trends associated with different stages of IGC progression. As the disease progressed, ALB levels and the A/G ratio declined overall, reaching the lowest values in the GC group, while renal function indicators (including UREA and CREA) gradually increased, peaking in the GC group. UA levels increased in the early disease stages but decreased in the GC group. HDL levels were significantly lower in the disease groups compared with the HCs. Peripheral blood cell parameters also showed stage-related changes. As disease severity increased, NEUT# and MONO# progressively increased, reaching their highest values in the GC group, whereas LYMPH counts were relatively higher in the precancerous stages but decreased in the GC group. RBC counts were significantly reduced in the GC group, while MCH and MCHC showed mild increasing trends. PLT was relatively elevated in the GC group, while MPV and PDW decreased with disease progression; all these differences were statistically significant. The detailed number of patients in each group and the characteristics of the respective indicators for the 2 external validation cohorts are provided in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendices 4</xref> and <xref ref-type="supplementary-material" rid="app5">5</xref>.</p></sec><sec id="s3-2"><title>Data Preprocessing and Validity Assessment</title><p>To address missing values, 6 imputation methods were applied, and the effects on feature distributions were systematically compared. Using the A/G index as an example, the boxplot results showed similar medians and IQRs across all imputation methods, with no obvious systematic shifts observed (<xref ref-type="fig" rid="figure2">Figure 2A</xref>). Further density distribution analyses showed that mean and median imputation smoothed the distributions to some extent, whereas linear, multiple, and regression imputation introduced varying degrees of adjustment to the local density structure (<xref ref-type="fig" rid="figure2">Figure 2B</xref>). In contrast, the distribution obtained by KNN imputation was highly consistent with the original data, fully preserving the pronounced bimodal structure and introducing almost no additional artifacts. Therefore, KNN imputation was selected as the optimal method, and the KNN-imputed dataset was used in all subsequent analyses. After completing missing-value imputation, the effects of the original data and SMOTE-augmented data on model performance were further compared. Compared with the nonaugmented data, the model accuracy after SMOTE augmentation increased from approximately 63% to about 80%, indicating that this approach effectively improves the model&#x2019;s ability to identify minority-class samples (<xref ref-type="fig" rid="figure2">Figure 2C</xref>). To assess whether data augmentation introduced potential distributional shifts, a comparative analysis of feature distributions between the original data and the SMOTE-augmented data was conducted (<xref ref-type="fig" rid="figure2">Figure 2D</xref>). The results demonstrate that the augmented data remained highly consistent with the original data in overall distribution shape, central tendency, and value range, with no apparent distributional distortion observed. Considering the potential presence of multicollinearity among clinical indicators, correlations among features were further calculated and visualized using a feature correlation heatmap (<xref ref-type="fig" rid="figure2">Figure 2E</xref>). Subsequently, the effects of different correlation coefficient thresholds on model performance were evaluated (<xref ref-type="fig" rid="figure2">Figure 2F</xref>), and the detailed performance metrics for each threshold are provided in <xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref>. The results indicate that when the correlation coefficient threshold was set to 0.65, the overall model performance reached its optimum, with all evaluation metrics stably maintained at approximately 80%, while retaining 27 features. Therefore, subsequent model construction and analyses were based on this selected feature set.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Preprocessing performance evaluation. (A) Box plot comparison of different imputation methods. (B) Density distribution comparison of different imputation methods. (C) Performance comparison before and after SMOTE augmentation. (D) Distribution of SMOTE-augmented data. (E) Feature correlation heatmap. (F) Impact of different feature selection thresholds on model performance. A/G: albumin/globulin ratio; KNN: k-nearest neighbor; SMOTE: synthetic minority oversampling technique; TP: true positive.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e94837_fig02.png"/></fig></sec><sec id="s3-3"><title>Model Comparison and External Validation</title><p>The results show that CatBoost performed the best on the internal training data (<xref ref-type="table" rid="table2">Table 2</xref>), achieving the highest accuracy of 0.9986 (95% CI 0.9978&#x2010;0.9993), sensitivity of 0.9997 (95% CI 0.9994&#x2010;0.9998), and specificity of 0.9986 (95% CI 0.9977&#x2010;0.9993). However, the mean accuracy across 5-fold cross-validation was 0.8087 (SD 0.0024), indicating overfitting and suggesting that the performance may be lower in actual applications. In contrast, AdaBoost performed the worst on the internal training data, with an accuracy of 0.5655 (95% CI 0.5543&#x2010;0.5765), which may reflect its limited ability to handle strong correlations among features. In the internal validation data, this overfitting phenomenon was further validated. Although its performance decreased, CatBoost still showed the best performance, with an accuracy of 0.8091 (95% CI 0.7972&#x2010;0.8211), specificity of 0.9527 (95% CI 0.9496&#x2010;0.9559), and sensitivity of 0.7857 (95% CI 0.7743&#x2010;0.7968), demonstrating ideal results. Other models, such as LGBM, also performed well in the internal validation, achieving an accuracy of 0.8067 (95% CI 0.7933&#x2010;0.8195), which is similar to that of CatBoost. During the external validation stage, data from GDPH and SOCH were used for initial transportability assessments to evaluate the model&#x2019;s performance in independent cohorts. Although CatBoost showed some fluctuations in these 2 external validation cohorts, the accuracy for the GDPH cohort was 0.7996 (95% CI 0.6800&#x2010;0.9000), sensitivity was 0.7682 (95% CI 0.6866&#x2010;0.8518), and specificity was 0.9314 (95% CI 0.8877&#x2010;0.9678); for the SOCH cohort, the accuracy was 0.8337 (95% CI 0.7553&#x2010;0.9111), sensitivity was 0.8491 (95% CI 0.7812&#x2010;0.9082), and specificity was 0.9573 (95% CI 0.9351&#x2010;0.9768). The performance was similar to that in the internal validation data, providing preliminary evidence of model transportability for clinical application.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Comparison of the performance of different ML models on internal training and validation data for diagnosing different stages of the Correa cascade, and the performance of the best model (CatBoost<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>) on 2 external validation cohorts.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Algorithm</td><td align="left" valign="bottom">Accuracy (95% CI)</td><td align="left" valign="bottom">Precision (95% CI)</td><td align="left" valign="bottom">Sensitivity (95% CI)</td><td align="left" valign="bottom">Specificity (95% CI)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (95% CI)</td><td align="char" char="hyphen" valign="bottom">5-Fold CV<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup>, mean (SD)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="7">Internal training</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>CatBoost</td><td align="left" valign="top">0.9986 (0.9978&#x2010;0.9993)</td><td align="left" valign="top">0.9984 (0.9974&#x2010;0.9993)</td><td align="left" valign="top">0.9997 (0.9994&#x2010;0.9998)</td><td align="left" valign="top">0.9986 (0.9977&#x2010;0.9993)</td><td align="left" valign="top">0.9985 (0.9975&#x2010;0.9993)</td><td align="left" valign="top">0.8087 (0.0024)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LGBM<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td><td align="left" valign="top">0.9986 (0.9977&#x2010;0.9995)</td><td align="left" valign="top">0.9986 (0.9977&#x2010;0.9994)</td><td align="left" valign="top">0.9984 (0.9974&#x2010;0.9993)</td><td align="left" valign="top">0.9997 (0.9994&#x2010;0.9999)</td><td align="left" valign="top">0.9985 (0.9975&#x2010;0.9994)</td><td align="left" valign="top">0.8065 (0.0083)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RF<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="top">0.9986 (0.9977&#x2010;0.9993)</td><td align="left" valign="top">0.9986 (0.9977&#x2010;0.9994)</td><td align="left" valign="top">0.9984 (0.9972&#x2010;0.9993)</td><td align="left" valign="top">0.9997 (0.9994&#x2010;0.9998)</td><td align="left" valign="top">0.9985 (0.9975&#x2010;0.9994)</td><td align="left" valign="top">0.7980 (0.0000)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>XGBoost<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup></td><td align="left" valign="top">0.9985 (0.9975&#x2010;0.9993)</td><td align="left" valign="top">0.9984 (0.9974&#x2010;0.9993)</td><td align="left" valign="top">0.9983 (0.9971&#x2010;0.9992)</td><td align="left" valign="top">0.9996 (0.9994&#x2010;0.9998)</td><td align="left" valign="top">0.9984 (0.9973&#x2010;0.9993)</td><td align="left" valign="top">0.7854 (0.0074)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DT<sup><xref ref-type="table-fn" rid="table2fn6">f</xref></sup></td><td align="left" valign="top">0.9891 (0.9865&#x2010;0.9915)</td><td align="left" valign="top">0.9895 (0.9871&#x2010;0.9918)</td><td align="left" valign="top">0.9884 (0.9858&#x2010;0.9911)</td><td align="left" valign="top">0.9972 (0.9966&#x2010;0.9978)</td><td align="left" valign="top">0.9942 (0.9924&#x2010;0.9959)</td><td align="left" valign="top">0.6356 (0.0123)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>AdaBoost<sup><xref ref-type="table-fn" rid="table2fn7">g</xref></sup></td><td align="left" valign="top">0.5655 (0.5543&#x2010;0.5765)</td><td align="left" valign="top">0.5550 (0.5441&#x2010;0.5661)</td><td align="left" valign="top">0.5533 (0.5427&#x2010;0.5637)</td><td align="left" valign="top">0.8912 (0.8884&#x2010;0.8940)</td><td align="left" valign="top">0.5531 (0.5427&#x2010;0.5639)</td><td align="left" valign="top">0.4971 (0.0047)</td></tr><tr><td align="left" valign="top" colspan="7">Internal validation</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>CatBoost</td><td align="left" valign="top">0.8091 (0.7972&#x2010;0.8211)</td><td align="left" valign="top">0.7806 (0.7675&#x2010;0.7934)</td><td align="left" valign="top">0.7857 (0.7743&#x2010;0.7968)</td><td align="left" valign="top">0.9527 (0.9496&#x2010;0.9559)</td><td align="left" valign="top">0.8005 (0.7881&#x2010;0.8127)</td><td align="left" valign="top">0.8087 (0.0024<bold>)</bold></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LGBM</td><td align="left" valign="top">0.8067 (0.7933&#x2010;0.8195)</td><td align="left" valign="top">0.7824 (0.7698&#x2010;0.7950)</td><td align="left" valign="top">0.7860 (0.7743&#x2010;0.7978)</td><td align="left" valign="top">0.9524 (0.9489&#x2010;0.9557)</td><td align="left" valign="top">0.7837 (0.7712&#x2010;0.7956)</td><td align="left" valign="top">0.8065 (0.0083)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RF</td><td align="left" valign="top">0.7930 (0.7796&#x2010;0.8063)</td><td align="left" valign="top">0.7615 (0.7466&#x2010;0.7772)</td><td align="left" valign="top">0.7679 (0.7564&#x2010;0.7806)</td><td align="left" valign="top">0.9485 (0.9451&#x2010;0.9518)</td><td align="left" valign="top">0.7538 (0.7404&#x2010;0.7664)</td><td align="left" valign="top">0.7980 (0.0000)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>XGBoost</td><td align="left" valign="top">0.7857 (0.7727&#x2010;0.7989)</td><td align="left" valign="top">0.7572 (0.7432&#x2010;0.7705)</td><td align="left" valign="top">0.7638 (0.7516&#x2010;0.7752)</td><td align="left" valign="top">0.9469 (0.9435&#x2010;0.9503)</td><td align="left" valign="top">0.7587 (0.7457&#x2010;0.7711)</td><td align="left" valign="top">0.7854 (0.0074)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DT</td><td align="left" valign="top">0.6344 (0.6183&#x2010;0.6491)</td><td align="left" valign="top">0.6098 (0.5945&#x2010;0.6251)</td><td align="left" valign="top">0.6163 (0.6017&#x2010;0.6307)</td><td align="left" valign="top">0.9091 (0.9050&#x2010;0.9128)</td><td align="left" valign="top">0.6112 (0.5966&#x2010;0.6264)</td><td align="left" valign="top">0.6356 (0.0123)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>AdaBoost</td><td align="left" valign="top">0.4974 (0.4817&#x2010;0.5131)</td><td align="left" valign="top">0.4872 (0.4721&#x2010;0.5031)</td><td align="left" valign="top">0.4854 (0.4702&#x2010;0.5007)</td><td align="left" valign="top">0.8742 (0.8701&#x2010;0.8782)</td><td align="left" valign="top">0.4854 (0.4702&#x2010;0.5007)</td><td align="left" valign="top">0.4971 (0.0047)</td></tr><tr><td align="left" valign="top" colspan="7">External validation (CatBoost)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GDPH<sup><xref ref-type="table-fn" rid="table2fn8">h</xref></sup></td><td align="left" valign="top">0.7996 (0.6800&#x2010;0.9000)</td><td align="left" valign="top">0.7996 (0.6800&#x2010;0.9000)</td><td align="left" valign="top">0.7682 (0.6866&#x2010;0.8518)</td><td align="left" valign="top">0.9314 (0.8877&#x2010;0.9678)</td><td align="left" valign="top">0.7889 (0.7003&#x2010;0.8703)</td><td align="left" valign="top">N/A<sup><xref ref-type="table-fn" rid="table2fn9">i</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>SOCH<sup><xref ref-type="table-fn" rid="table2fn10">j</xref></sup></td><td align="left" valign="top">0.8337 (0.7553&#x2010;0.9111)</td><td align="left" valign="top">0.8337 (0.7553&#x2010;0.9111)</td><td align="left" valign="top">0.8491 (0.7812&#x2010;0.9082)</td><td align="left" valign="top">0.9573 (0.9351&#x2010;0.9768)</td><td align="left" valign="top">0.8439 (0.7713&#x2010;0.9063)</td><td align="left" valign="top">N/A</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>CatBoost: categorical boosting.</p></fn><fn id="table2fn2"><p><sup>b</sup>CV: cross-validation.</p></fn><fn id="table2fn3"><p><sup>c</sup>LGBM: light gradient boosting machine.</p></fn><fn id="table2fn4"><p><sup>d</sup>RF: random forest.</p></fn><fn id="table2fn5"><p><sup>e</sup>XGBoost: extreme gradient boosting.</p></fn><fn id="table2fn6"><p><sup>f</sup>DT: decision tree.</p></fn><fn id="table2fn7"><p><sup>g</sup>AdaBoost: adaptive boosting.</p></fn><fn id="table2fn8"><p><sup>h</sup>GDPH: Guangdong Provincial People&#x2019;s Hospital.</p></fn><fn id="table2fn9"><p><sup>i</sup>Not applicable.</p></fn><fn id="table2fn10"><p><sup>j</sup>SOCH: Shengli Oilfield Central Hospital.</p></fn></table-wrap-foot></table-wrap><p>In external validation, the CatBoost model demonstrated good discriminative performance across 2 independent centers. In the GDPH external validation cohort, the ROC curve (area under the curve [AUC]) was 0.94 (<xref ref-type="fig" rid="figure3">Figure 3A</xref>), while the AUC of the SOCH external validation cohort was 0.97 (<xref ref-type="fig" rid="figure3">Figure 3B</xref>), indicating promising discriminative ability of the model at both centers. The confusion matrices illustrated the detailed classification performance across different disease stages. In the GDPH cohort, the correct classification rate for AG was 87%, with approximately 13% of AG samples misclassified into the adjacent IM stage (<xref ref-type="fig" rid="figure3">Figure 3C</xref>). The IM category showed the best recognition performance, achieving a 100% correct classification rate. The correct classification rate for GC was 70%, with the remaining 30% mainly misclassified as AG (20%) and IM (10%). In the SOCH cohort, the classification accuracies for HC and GC both reached 100% (<xref ref-type="fig" rid="figure3">Figure 3D</xref>). The correct classification rates for NAG and AG were 80% and 90%, respectively, with a small number of samples showing cross-misclassification between NAG and AG. The correct classification rate for IM was 65%, with the remaining 35% misclassified as NAG. Misclassifications in this cohort were concentrated in intermediate pathological stages, suggesting that the model was capable of identifying extreme states (HC and GC). The consistency of probability predictions in the external validation cohorts was assessed using calibration curves and Brier scores. In the GDPH cohort (<xref ref-type="fig" rid="figure3">Figure 3E</xref>), the calibration curve closely approximated the ideal reference line, with a Brier score of 0.0257. In the SOCH cohort, the Brier score was 0.0830 (<xref ref-type="fig" rid="figure3">Figure 3F</xref>), indicating good agreement between predicted probabilities and observed outcomes. DCA demonstrated that the CatBoost model provided potential clinical utility in both external validation centers. In the GDPH cohort (<xref ref-type="fig" rid="figure3">Figure 3G</xref>), when the threshold probability ranged from approximately 0.15 to 0.9, the net benefit of the model consistently exceeded that of the &#x201C;treat-all&#x201D; and &#x201C;treat-none&#x201D; strategies, with stable positive gains observed in the low-to-moderate threshold range. In the SOCH cohort (<xref ref-type="fig" rid="figure3">Figure 3H</xref>), the model maintained a significant net benefit advantage across a wider threshold probability range (approximately 0.02&#x2010;0.98), suggesting its potential net benefit advantage. Collectively, these findings suggest that the CatBoost model has the potential to provide effective support for clinical diagnosis.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Performance evaluation of CatBoost (categorical boosting) in the external validation cohort. ROC curve in the (A) GDPH and (B) SOCH external validation cohort. Confusion matrix in the (C) GDPH and (D) SOCH external validation cohort. Calibration curve in the (E) GDPH and (F) SOCH external validation cohort. DCA in the (G) GDPH and (H) SOCH external validation cohort. AG: atrophic gastritis; DCA: decision curve analysis; GC: gastric cancer; GDPH: Guangdong Provincial People&#x2019;s Hospital; HC: healthy control; IM: intestinal metaplasia; NAG: nonatrophic gastritis; ROC: receiver operating characteristic; SOCH: Shengli Oilfield Central Hospital.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e94837_fig03.png"/></fig></sec><sec id="s3-4"><title>Model Interpretation</title><p>SHAP explainability analysis is used to gain further insight into the model&#x2019;s decision-making process. <xref ref-type="fig" rid="figure4">Figure 4A</xref> presents the top 20 most important features in the model&#x2019;s decision-making process, quantifying the significance of these features at different stages. For instance, MONO% plays a crucial role at the IM and GC stages, while DBIL is particularly important at the GC stage. <xref ref-type="fig" rid="figure4">Figure 4B</xref> shows how the top 10 features influence the decision boundary of the predicted outcomes. The results reveal that different features exhibit distinct threshold effects across different disease stages. For example, the threshold range for MONO% is between 3 and 7.5, which closely aligns with its clinical reference range (3-10), while the threshold range for DBIL is between 2 and 6, also close to its clinical reference range (0&#x2010;6.8). These findings suggest that diagnostic models built on these indicators have the potential to assist in disease screening and staging. The density distribution plot in <xref ref-type="fig" rid="figure4">Figure 4C</xref> further illustrates the distribution of the top 10 features across different stages. For example, indicators such as MONO%, BASO%, and NEUT# show significant differences in distribution and intensity between the HC and GC groups, confirming that these features play a crucial role in distinguishing between different disease stages. Additionally, since age is among the top 10 most important features, we further evaluated the model&#x2019;s performance across different age strata in 2 external validation cohorts. The results showed that the CatBoost model continued to effectively differentiate patient data across various age strata. Detailed results are provided in <xref ref-type="supplementary-material" rid="app7">Multimedia Appendix 7</xref>.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Global model interpretation using the SHAP method. (A) SHAP summary bar plot. (B) SHAP dependence plots for the top 10 features. Each dependence plot shows how a single feature affects the output of the prediction model, and each dot represents a single patient. (C) Density distributions of the top 10 features at different stages. A/G: albumin/globulin ratio; AG: atrophic gastritis; AKP: alkaline phosphatase; ALB: albumin; AST: aspartate aminotransferase; BASO%: basophil percentage; CREA: creatinine; DBIL: total bilirubin; EO#: eosinophil count; GC: gastric cancer; HC: healthy control; IM: intestinal metaplasia; LYMPH#: lymphocyte count; MONO#: monocyte count; MONO%: monocyte percentage; NAG: nonatrophic gastritis; NEUT#: neutrophil count; PDW: platelet distribution width; PLT: platelet count; SHAP: Shapley Additive Explanations; UA: uric acid; UREA: urea.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e94837_fig04.png"/></fig></sec><sec id="s3-5"><title>Web Server Development</title><p>To maximize efficient use of the model and enhance its translational potential, we embedded a model constructed from the top 10 features ranked by SHAP into the Streamlit platform (<xref ref-type="fig" rid="figure5">Figure 5</xref>). Users need to enter only the 10 required features, and the application will automatically predict the stage of gastric disease progression for a given patient. This web tool is freely accessible through direct access to the Streamlit platform [<xref ref-type="bibr" rid="ref27">27</xref>].</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Web page tool presentation of the optimal CatBoost model. The final model includes 10 features for predicting different stages. After the 10 features are entered, the prediction result will be automatically displayed on the right side. A/G: albumin/globulin ratio; AST: aspartate aminotransferase; BASO: basophils; CatBoost: categorical boosting; CREA: creatinine; DBIL: total bilirubin; LYMPH#: lymphocyte count; NAG: nonatrophic gastritis; NEUT#: neutrophil count; PDW: platelet distribution width; UA: uric acid.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e94837_fig05.png"/></fig></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This study developed and validated an explainable ML-based model for predicting different stages in the progression of <italic>H pylori</italic>&#x2013;initiated IGC. The model demonstrated robust performance across internal validation and external cohorts, suggesting that the ML-based model may serve as a potential tool for IGC staging. It supports early screening and facilitates informed clinical decision-making for subsequent evaluation and treatment.</p></sec><sec id="s4-2"><title>Comparison With Prior Work</title><p>Currently, most diagnostic models for gastric diseases based on routine laboratory indicators focus on binary classification at a single disease stage. Previous studies have primarily focused on discriminating between IM and NAG [<xref ref-type="bibr" rid="ref28">28</xref>], the prediction of future GC risk in initially negative individuals [<xref ref-type="bibr" rid="ref29">29</xref>], the assessment of laboratory parameter changes before and after treatment in patients with GC [<xref ref-type="bibr" rid="ref30">30</xref>], or differentiation between GC and precancerous lesions [<xref ref-type="bibr" rid="ref31">31</xref>]. In contrast, dynamic changes in hematological parameters across different stages of <italic>H pylori&#x2013;</italic>initiated IGC remain poorly understood, as well as the feasibility of using these parameters for multistage prediction models also remains unclear.</p><p>In this study, we constructed a multiclass ML model for predicting different developmental stages of IGC. Among the evaluated models, CatBoost achieved the best diagnostic performance, with an accuracy of 0.8091 in the internal validation cohort and accuracies of 0.7996 and 0.8337 in 2 independent test sets, respectively. DCA further supported its potential clinical applicability. However, the relatively large number of features included in the current model may limit its practical implementation. Therefore, it will be important to develop more lightweight models and identify a more concise set of key features while maintaining predictive performance.</p><p>Given the lack of unified guidelines for feature selection in predictive models, SHAP analysis was used to provide both global and local explanations of the model, thereby elucidating the contribution of individual features to the model&#x2019;s predictions [<xref ref-type="bibr" rid="ref32">32</xref>]. This study demonstrates that routine laboratory indicators, including MONO%, A/G, BASO%, PDW, and DBIL, play important roles in distinguishing different disease stages. Higher MONO% was associated with an increased predicted risk of GC, which is consistent with previous findings showing that elevated monocyte proportion is associated with poorer prognosis and reduced overall survival [<xref ref-type="bibr" rid="ref33">33</xref>]. In contrast, lower A/G levels markedly increased the likelihood of GC classification, in agreement with epidemiological evidence linking low A/G levels to increased mortality across multiple malignancies, including GC [<xref ref-type="bibr" rid="ref34">34</xref>]. In addition, age was ranked among the top 10 most important features. Given that the incidence of GC increases significantly with age [<xref ref-type="bibr" rid="ref35">35</xref>], we further evaluated the model&#x2019;s predictive performance across different age strata. The results demonstrated that the CatBoost model was able to consistently identify patients at different disease stages across age groups. Furthermore, based on the top 10 key features and the Streamlit framework, we developed a user-friendly online prediction platform to enhance its clinical accessibility.</p></sec><sec id="s4-3"><title>Study Limitations</title><p>This study has the following limitations. First, it was based on populations at 2 hospitals in China, which may have been influenced by factors such as demographic structure and lifestyle habits. Therefore, the generalizability of the results needs further validation. However, the study provides preliminary evidence for diagnosing IGC based on routine laboratory hematological indicators. Second, although the model can distinguish between different stages of IGC progression to some extent, its relatively high specificity and slightly lower sensitivity may lead to missed or incorrect classifications, especially in late-stage patients, which is unacceptable for practical applications. Therefore, the current method should be used only as an auxiliary diagnostic tool and should be combined with routine clinical methods. Additionally, the construction of ML models requires sufficiently large, balanced, and representative clinical datasets. This study exhibits biases in the HC, AG, and GC groups. Although clinical data augmentation using the SMOTE method has been widely applied [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>], its assumption of a uniform feature space means the synthetic data generated may not fully reflect real clinical heterogeneity or maintain complete clinical interpretability, potentially increasing the risk of overfitting. The relatively small external validation cohorts may also be insufficient to comprehensively assess the model&#x2019;s generalizability. Therefore, future research should expand the sample size and include data from more collaborative centers to enhance the model&#x2019;s generalizability. Moreover, the testing platforms differ across hospitals and may affect routine laboratory indicators and thus may influence model transferability. Although the study did not demonstrate the specific effects of platform differences on diagnostic outcomes in the results section, interinstrument and interlaboratory variability remains an important practical barrier to broader application. Therefore, promoting mutual recognition of test results and standardization of clinical reference ranges at regional and national levels is crucial. Finally, this study focuses only on IGC caused by <italic>H pylori</italic> infection, and its applicability to other types of GC, such as diffuse GC or signet-ring cell carcinoma, has not been clarified. Future studies should expand the scope of research to evaluate the prospects of using routine laboratory indicators in different types of GC.</p></sec><sec id="s4-4"><title>Conclusions</title><p>In conclusion, this study developed and validated an interpretable ML model based on routine clinical laboratory data to predict different stages in the progression of <italic>H pylori</italic> infection-associated IGC. The model demonstrated promising predictive performance in the internal validation and provided preliminary evidence of transportability in 2 external cohorts. As a potential auxiliary diagnostic tool, it shows promising applicability in remote areas and health care settings with limited medical resources, helping to optimize resource allocation and support early clinical intervention.</p></sec></sec></body><back><ack><p>The authors declare the use of generative AI during manuscript preparation. According to the Generative Artificial Intelligence Delegation Taxonomy (GAIDeT, 2025), the following tasks were performed with the assistance of generative AI tools under full human supervision: language polishing and wording refinement. The tool used was ChatGPT by OpenAI. Responsibility for the final manuscript lies entirely with the authors. All AI-assisted text was reviewed and revised by the authors before submission. Generative AI tools are not listed as authors and do not bear responsibility for the final outcomes.</p></ack><notes><sec><title>Funding</title><p>This study was financially supported by the Research Foundation for Advanced Talents of Guangdong Provincial People&#x2019;s Hospital (grant KY012023293) and the Young Top-Talent in Science and Technology Innovation of the Guangdong Special Support Program (grant 2025TQ09A269). JT acknowledges the support of the Research Training Program scholarship by the Australian Commonwealth Government.</p></sec><sec><title>Data Availability</title><p>Data collected for the study, including deidentified individual participant data and a data dictionary defining each field in the dataset, will be made available upon request to the corresponding author. The model and scripts developed in this study are available at GitHub [<xref ref-type="bibr" rid="ref36">36</xref>].</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: LW</p><p>Data curation: JT, HC</p><p>Formal analysis: JT, HC</p><p>Funding acquisition: JT, LW</p><p>Investigation: JT, WZ</p><p>Methodology: JT</p><p>Project administration: CM, LW</p><p>Resources: CM, LW, HC</p><p>Supervision: CM, LW, ACYT, BJM</p><p>Validation: JT, WZ, LW</p><p>Visualization: JT</p><p>Writing &#x2013; original draft: JT, HC, WZ, BJM, ACYT, CM, LW</p><p>Writing &#x2013; review &#x0026; editing: JT, HC, WZ, BJM, ACYT, CM, LW</p><p>JT and HC shared first authorship and contributed equally to the study. CM shared senior authorship with LW. CM is co-corresponding author of the study. All authors have read and agreed to the published version of this paper.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">A/G</term><def><p>albumin/globulin ratio</p></def></def-item><def-item><term id="abb2">AdaBoost</term><def><p>adaptive boosting</p></def></def-item><def-item><term id="abb3">AG</term><def><p>atrophic gastritis</p></def></def-item><def-item><term id="abb4">AKP</term><def><p>alkaline phosphatase</p></def></def-item><def-item><term id="abb5">ALB</term><def><p>albumin</p></def></def-item><def-item><term id="abb6">AST</term><def><p>aspartate aminotransferase</p></def></def-item><def-item><term id="abb7">AUC</term><def><p>area under the curve</p></def></def-item><def-item><term id="abb8">BASO%</term><def><p>basophil percentage</p></def></def-item><def-item><term id="abb9">CatBoost</term><def><p>categorical boosting</p></def></def-item><def-item><term id="abb10">CREA</term><def><p>creatinine</p></def></def-item><def-item><term id="abb11">CRP</term><def><p>C-reactive protein</p></def></def-item><def-item><term id="abb12">CV</term><def><p>cross-validation</p></def></def-item><def-item><term id="abb13">DBIL</term><def><p>total bilirubin</p></def></def-item><def-item><term id="abb14">DCA</term><def><p>decision curve analysis</p></def></def-item><def-item><term id="abb15">DT</term><def><p>decision tree</p></def></def-item><def-item><term id="abb16">EO#</term><def><p>eosinophil count</p></def></def-item><def-item><term id="abb17">GC</term><def><p>gastric cancer</p></def></def-item><def-item><term id="abb18">GDPH</term><def><p>Guangdong Provincial People&#x2019;s Hospital</p></def></def-item><def-item><term id="abb19">GGT</term><def><p>gamma-glutamyl transferase</p></def></def-item><def-item><term id="abb20">GLU</term><def><p>glucose</p></def></def-item><def-item><term id="abb21">HC</term><def><p>healthy control</p></def></def-item><def-item><term id="abb22">HDL</term><def><p>high-density lipoprotein cholesterol</p></def></def-item><def-item><term id="abb23">IARC</term><def><p>International Agency for Research on Cancer</p></def></def-item><def-item><term id="abb24">IGC</term><def><p>intestinal-type gastric cancer</p></def></def-item><def-item><term id="abb25">IL-6</term><def><p>interleukin-6</p></def></def-item><def-item><term id="abb26">IM</term><def><p>intestinal metaplasia</p></def></def-item><def-item><term id="abb27">KNN</term><def><p>k-nearest neighbor</p></def></def-item><def-item><term id="abb28">LGBM</term><def><p>light gradient boosting machine</p></def></def-item><def-item><term id="abb29">LIS</term><def><p>laboratory information systems</p></def></def-item><def-item><term id="abb30">LYMPH#</term><def><p>lymphocyte count</p></def></def-item><def-item><term id="abb31">MCH</term><def><p>mean corpuscular hemoglobin</p></def></def-item><def-item><term id="abb32">MCHC</term><def><p>mean corpuscular hemoglobin concentration</p></def></def-item><def-item><term id="abb33">ML</term><def><p>machine learning</p></def></def-item><def-item><term id="abb34">MONO%</term><def><p>monocyte percentage</p></def></def-item><def-item><term id="abb35">MPV</term><def><p>mean platelet volume</p></def></def-item><def-item><term id="abb36">NAG</term><def><p>nonatrophic gastritis</p></def></def-item><def-item><term id="abb37">NCC</term><def><p>National Cancer Center of China</p></def></def-item><def-item><term id="abb38">NEUT#</term><def><p>neutrophil count</p></def></def-item><def-item><term id="abb39">NLR</term><def><p>neutrophil-to-lymphocyte ratio</p></def></def-item><def-item><term id="abb40">PCT</term><def><p>procalcitonin</p></def></def-item><def-item><term id="abb41">PDW</term><def><p>platelet distribution width</p></def></def-item><def-item><term id="abb42">PLR</term><def><p>platelet-to-lymphocyte ratio</p></def></def-item><def-item><term id="abb43">RBC</term><def><p>red blood cell</p></def></def-item><def-item><term id="abb44">RDW</term><def><p>red cell distribution width</p></def></def-item><def-item><term id="abb45">RF</term><def><p>random forest</p></def></def-item><def-item><term id="abb46">ROC</term><def><p>receiver operating characteristic</p></def></def-item><def-item><term id="abb47">SHAP</term><def><p>Shapley Additive Explanations</p></def></def-item><def-item><term id="abb48">SMOTE</term><def><p>synthetic minority oversampling technique</p></def></def-item><def-item><term id="abb49">SOCH</term><def><p>Shengli Oilfield Central Hospital</p></def></def-item><def-item><term id="abb50">TC</term><def><p>total cholesterol</p></def></def-item><def-item><term id="abb51">TG</term><def><p>triglycerides</p></def></def-item><def-item><term id="abb52">UA</term><def><p>uric acid</p></def></def-item><def-item><term id="abb53">UREA</term><def><p>urea</p></def></def-item><def-item><term id="abb54">WBC</term><def><p>white blood cell</p></def></def-item><def-item><term id="abb55">XGBoost</term><def><p>extreme gradient boosting</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>P</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Teng</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Shu</surname><given-names>P</given-names> </name></person-group><article-title>Trends and projections of the burden of gastric cancer in China and G20 countries: a comparative study based on the Global Burden of Disease database 2021</article-title><source>Int J Surg</source><year>2025</year><month>07</month><day>1</day><volume>111</volume><issue>7</issue><fpage>4854</fpage><lpage>4865</lpage><pub-id pub-id-type="doi">10.1097/JS9.0000000000002464</pub-id><pub-id pub-id-type="medline">40359560</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Han</surname><given-names>B</given-names> </name><name name-style="western"><surname>Zheng</surname><given-names>R</given-names> </name><name name-style="western"><surname>Zeng</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Cancer incidence and mortality in China, 2022</article-title><source>J Natl Cancer Cent</source><year>2024</year><month>03</month><volume>4</volume><issue>1</issue><fpage>47</fpage><lpage>53</lpage><pub-id pub-id-type="doi">10.1016/j.jncc.2024.01.006</pub-id><pub-id pub-id-type="medline">39036382</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Correa</surname><given-names>P</given-names> </name></person-group><article-title>Human gastric carcinogenesis: a multistep and multifactorial process&#x2014;First American Cancer Society Award Lecture on Cancer Epidemiology and Prevention</article-title><source>Cancer Res</source><year>1992</year><month>12</month><day>15</day><volume>52</volume><issue>24</issue><fpage>6735</fpage><lpage>6740</lpage><pub-id pub-id-type="medline">1458460</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cook</surname><given-names>KW</given-names> </name><name name-style="western"><surname>Letley</surname><given-names>DP</given-names> </name><name name-style="western"><surname>Ingram</surname><given-names>RJM</given-names> </name><etal/></person-group><article-title>CCL20/CCR6-mediated migration of regulatory T cells to the Helicobacter pylori-infected human gastric mucosa</article-title><source>Gut</source><year>2014</year><month>10</month><volume>63</volume><issue>10</issue><fpage>1550</fpage><lpage>1559</lpage><pub-id pub-id-type="doi">10.1136/gutjnl-2013-306253</pub-id><pub-id pub-id-type="medline">24436142</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>G</given-names> </name><name name-style="western"><surname>Li</surname><given-names>G</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Deep learning-aided optical biopsy achieves whole-chain diagnosis of Correa cascade of gastric cancer: a prospective study</article-title><source>BMC Med</source><year>2025</year><month>09</month><day>30</day><volume>23</volume><issue>1</issue><fpage>527</fpage><pub-id pub-id-type="doi">10.1186/s12916-025-04310-9</pub-id><pub-id pub-id-type="medline">41029674</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Unveiling gastric precancerous stages: metabolomic insights for early detection and intervention</article-title><source>BMC Gastroenterol</source><year>2025</year><month>04</month><day>29</day><volume>25</volume><issue>1</issue><fpage>318</fpage><pub-id pub-id-type="doi">10.1186/s12876-025-03898-9</pub-id><pub-id pub-id-type="medline">40301782</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pimentel-Nunes</surname><given-names>P</given-names> </name><name name-style="western"><surname>Dinis-Ribeiro</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ponchon</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Endoscopic submucosal dissection: European Society of Gastrointestinal Endoscopy (ESGE) guideline</article-title><source>Endoscopy</source><year>2015</year><month>09</month><volume>47</volume><issue>9</issue><fpage>829</fpage><lpage>854</lpage><pub-id pub-id-type="doi">10.1055/s-0034-1392882</pub-id><pub-id pub-id-type="medline">26317585</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pimentel-Nunes</surname><given-names>P</given-names> </name><name name-style="western"><surname>Lib&#x00E2;nio</surname><given-names>D</given-names> </name><name name-style="western"><surname>Marcos-Pinto</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Management of epithelial precancerous conditions and lesions in the stomach (MAPS II): European Society of Gastrointestinal Endoscopy (ESGE), European Helicobacter and Microbiota Study Group (EHMSG), European Society of Pathology (ESP), and Sociedade Portuguesa de Endoscopia Digestiva (SPED) guideline update 2019</article-title><source>Endoscopy</source><year>2019</year><month>04</month><volume>51</volume><issue>4</issue><fpage>365</fpage><lpage>388</lpage><pub-id pub-id-type="doi">10.1055/a-0859-1883</pub-id><pub-id pub-id-type="medline">30841008</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Si</surname><given-names>YT</given-names> </name><name name-style="western"><surname>Xiong</surname><given-names>XS</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>JT</given-names> </name><etal/></person-group><article-title>Identification of chronic non-atrophic gastritis and intestinal metaplasia stages in the Correa&#x2019;s cascade through machine learning analyses of SERS spectral signature of non-invasively-collected human gastric fluid samples</article-title><source>Biosens Bioelectron</source><year>2024</year><month>10</month><day>15</day><volume>262</volume><fpage>116530</fpage><pub-id pub-id-type="doi">10.1016/j.bios.2024.116530</pub-id><pub-id pub-id-type="medline">38943854</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dai</surname><given-names>D</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Interactions between gastric microbiota and metabolites in gastric cancer</article-title><source>Cell Death Dis</source><year>2021</year><month>11</month><day>24</day><volume>12</volume><issue>12</issue><fpage>1104</fpage><pub-id pub-id-type="doi">10.1038/s41419-021-04396-y</pub-id><pub-id pub-id-type="medline">34819503</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Tian</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhan</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Accurate diagnosis and prognosis prediction of gastric cancer using deep learning on digital pathological images: a retrospective multicentre study</article-title><source>EBioMedicine</source><year>2021</year><month>11</month><volume>73</volume><fpage>103631</fpage><pub-id pub-id-type="doi">10.1016/j.ebiom.2021.103631</pub-id><pub-id pub-id-type="medline">34678610</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Song</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zou</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>W</given-names> </name><etal/></person-group><article-title>Clinically applicable histopathological diagnosis system for gastric cancer detection using deep learning</article-title><source>Nat Commun</source><year>2020</year><month>08</month><day>27</day><volume>11</volume><issue>1</issue><fpage>4294</fpage><pub-id pub-id-type="doi">10.1038/s41467-020-18147-8</pub-id><pub-id pub-id-type="medline">32855423</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cheng</surname><given-names>S</given-names> </name><name name-style="western"><surname>Han</surname><given-names>F</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>The red distribution width and the platelet distribution width as prognostic predictors in gastric cancer</article-title><source>BMC Gastroenterol</source><year>2017</year><month>12</month><day>20</day><volume>17</volume><issue>1</issue><fpage>163</fpage><pub-id pub-id-type="doi">10.1186/s12876-017-0685-7</pub-id><pub-id pub-id-type="medline">29262773</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yoshida</surname><given-names>N</given-names> </name><name name-style="western"><surname>Kosumi</surname><given-names>K</given-names> </name><name name-style="western"><surname>Tokunaga</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Clinical importance of mean corpuscular volume as a prognostic marker after esophagectomy for esophageal cancer: a retrospective study</article-title><source>Ann Surg</source><year>2020</year><month>03</month><volume>271</volume><issue>3</issue><fpage>494</fpage><lpage>501</lpage><pub-id pub-id-type="doi">10.1097/SLA.0000000000002971</pub-id><pub-id pub-id-type="medline">29995687</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>B</given-names> </name><name name-style="western"><surname>Dai</surname><given-names>D</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Pretreatment hematocrit is superior to hemoglobin as a prognostic factor for triple negative breast cancer</article-title><source>PLoS One</source><year>2016</year><volume>11</volume><issue>11</issue><fpage>e0165133</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0165133</pub-id><pub-id pub-id-type="medline">27851755</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>D</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Li</surname><given-names>L</given-names> </name><name name-style="western"><surname>Song</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>W</given-names> </name></person-group><article-title>The neutrophil to lymphocyte ratio may predict benefit from chemotherapy in lung cancer</article-title><source>Cell Physiol Biochem</source><year>2018</year><volume>46</volume><issue>4</issue><fpage>1595</fpage><lpage>1605</lpage><pub-id pub-id-type="doi">10.1159/000489207</pub-id><pub-id pub-id-type="medline">29694985</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Duan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Li</surname><given-names>G</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>H</given-names> </name></person-group><article-title>Single and combined use of the platelet-lymphocyte ratio, neutrophil-lymphocyte ratio, and systemic immune-inflammation index in gastric cancer diagnosis</article-title><source>Front Oncol</source><year>2023</year><volume>13</volume><fpage>1143154</fpage><pub-id pub-id-type="doi">10.3389/fonc.2023.1143154</pub-id><pub-id pub-id-type="medline">37064093</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xin-Ji</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Yong-Gang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Xiao-Jun</surname><given-names>S</given-names> </name><name name-style="western"><surname>Xiao-Wu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Dong</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Da-Jian</surname><given-names>Z</given-names> </name></person-group><article-title>The prognostic role of neutrophils to lymphocytes ratio and platelet count in gastric cancer: a meta-analysis</article-title><source>Int J Surg</source><year>2015</year><month>09</month><volume>21</volume><issue>84-91</issue><fpage>84</fpage><lpage>91</lpage><pub-id pub-id-type="doi">10.1016/j.ijsu.2015.07.681</pub-id><pub-id pub-id-type="medline">26225826</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aliustaoglu</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bilici</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ustaalioglu</surname><given-names>BBO</given-names> </name><etal/></person-group><article-title>The effect of peripheral blood values on prognosis of patients with locally advanced gastric cancer before treatment</article-title><source>Med Oncol</source><year>2010</year><month>12</month><volume>27</volume><issue>4</issue><fpage>1060</fpage><lpage>1065</lpage><pub-id pub-id-type="doi">10.1007/s12032-009-9335-4</pub-id><pub-id pub-id-type="medline">19847679</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ilhan</surname><given-names>N</given-names> </name><name name-style="western"><surname>Ilhan</surname><given-names>N</given-names> </name><name name-style="western"><surname>Ilhan</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Akbulut</surname><given-names>H</given-names> </name><name name-style="western"><surname>Kucuksu</surname><given-names>M</given-names> </name></person-group><article-title>C-reactive protein, procalcitonin, interleukin-6, vascular endothelial growth factor and oxidative metabolites in diagnosis of infection and staging in patients with gastric cancer</article-title><source>World J Gastroenterol</source><year>2004</year><month>04</month><day>15</day><volume>10</volume><issue>8</issue><fpage>1115</fpage><lpage>1120</lpage><pub-id pub-id-type="doi">10.3748/wjg.v10.i8.1115</pub-id><pub-id pub-id-type="medline">15069709</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Iida</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ikeda</surname><given-names>F</given-names> </name><name name-style="western"><surname>Ninomiya</surname><given-names>T</given-names> </name><etal/></person-group><article-title>White blood cell count and risk of gastric cancer incidence in a general Japanese population: the Hisayama study</article-title><source>Am J Epidemiol</source><year>2012</year><month>03</month><day>15</day><volume>175</volume><issue>6</issue><fpage>504</fpage><lpage>510</lpage><pub-id pub-id-type="doi">10.1093/aje/kwr345</pub-id><pub-id pub-id-type="medline">22366378</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lai</surname><given-names>JX</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Gong</surname><given-names>SS</given-names> </name><etal/></person-group><article-title>Development and validation of an interpretable risk prediction model for the early classification of thalassemia</article-title><source>NPJ Digit Med</source><year>2025</year><month>06</month><day>10</day><volume>8</volume><issue>1</issue><fpage>346</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01766-0</pub-id><pub-id pub-id-type="medline">40494920</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tang</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Xiong</surname><given-names>XS</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>TT</given-names> </name><etal/></person-group><article-title>Rapid discrimination of Mycobacterium tuberculosis and non-tuberculous mycobacteria disease via interpretive machine learning analysis of routine laboratory tests</article-title><source>BMJ Health Care Inform</source><year>2025</year><month>10</month><day>17</day><volume>32</volume><issue>1</issue><fpage>e101575</fpage><pub-id pub-id-type="doi">10.1136/bmjhci-2025-101575</pub-id><pub-id pub-id-type="medline">41106844</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Cheng</surname><given-names>L</given-names> </name><name name-style="western"><surname>Niu</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Identification tool for gastric cancer based on integration of 33 clinical available blood indices through deep learning</article-title><source>IEEE Access</source><year>2022</year><volume>10</volume><fpage>106081</fpage><lpage>106092</lpage><pub-id pub-id-type="doi">10.1109/ACCESS.2022.3172477</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Xie</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Machine learning for predicting in-hospital mortality in elderly patients with heart failure combined with hypertension: a multicenter retrospective study</article-title><source>Cardiovasc Diabetol</source><year>2024</year><month>11</month><day>15</day><volume>23</volume><issue>1</issue><fpage>407</fpage><pub-id pub-id-type="doi">10.1186/s12933-024-02503-9</pub-id><pub-id pub-id-type="medline">39548495</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Choi</surname><given-names>BK</given-names> </name><name name-style="western"><surname>Choi</surname><given-names>YJ</given-names> </name><name name-style="western"><surname>Sung</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Development and validation of an artificial intelligence model for the early classification of the aetiology of meningitis and encephalitis: a retrospective observational study</article-title><source>EClinicalMedicine</source><year>2023</year><month>07</month><volume>61</volume><fpage>102051</fpage><pub-id pub-id-type="doi">10.1016/j.eclinm.2023.102051</pub-id><pub-id pub-id-type="medline">37415843</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="web"><article-title>Correa cascade prediction</article-title><source>Streamlit</source><access-date>2026-07-25</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://hp-igc.streamlit.app/">https://hp-igc.streamlit.app/</ext-link></comment></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Bi</surname><given-names>J</given-names> </name><name name-style="western"><surname>Song</surname><given-names>S</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Gong</surname><given-names>A</given-names> </name></person-group><article-title>Identifying gastric intestinal metaplasia risk based on clinical indicators: a machine learning predictive model based on the SHAP methodology</article-title><source>Front Pharmacol</source><year>2025</year><volume>16</volume><fpage>1602191</fpage><pub-id pub-id-type="doi">10.3389/fphar.2025.1602191</pub-id><pub-id pub-id-type="medline">41282630</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Taninaga</surname><given-names>J</given-names> </name><name name-style="western"><surname>Nishiyama</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Fujibayashi</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Prediction of future gastric cancer risk using a machine learning algorithm and comprehensive medical check-up data: a case-control study</article-title><source>Sci Rep</source><year>2019</year><month>08</month><day>27</day><volume>9</volume><issue>1</issue><fpage>12384</fpage><pub-id pub-id-type="doi">10.1038/s41598-019-48769-y</pub-id><pub-id pub-id-type="medline">31455831</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rafiepoor</surname><given-names>H</given-names> </name><name name-style="western"><surname>Banoei</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Ghorbankhanloo</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Exploring the potential of machine learning in gastric cancer: prognostic biomarkers, subtyping, and stratification</article-title><source>BMC Cancer</source><year>2025</year><month>04</month><day>30</day><volume>25</volume><issue>1</issue><fpage>809</fpage><pub-id pub-id-type="doi">10.1186/s12885-025-14204-x</pub-id><pub-id pub-id-type="medline">40307780</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ke</surname><given-names>X</given-names> </name><name name-style="western"><surname>Cai</surname><given-names>X</given-names> </name><name name-style="western"><surname>Bian</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Predicting early gastric cancer risk using machine learning: a population-based retrospective study</article-title><source>Digit Health</source><year>2024</year><volume>10</volume><fpage>20552076241240905</fpage><pub-id pub-id-type="doi">10.1177/20552076241240905</pub-id><pub-id pub-id-type="medline">38559579</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>ZZ</given-names> </name><name name-style="western"><surname>Yuan</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>YD</given-names> </name><etal/></person-group><article-title>Validation and interpretation of machine-learning models for rapid identification of active tuberculosis infection using routine laboratory indicators</article-title><source>Front Cell Infect Microbiol</source><year>2025</year><volume>15</volume><fpage>1718614</fpage><pub-id pub-id-type="doi">10.3389/fcimb.2025.1718614</pub-id><pub-id pub-id-type="medline">41488479</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Feng</surname><given-names>F</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zheng</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Low lymphocyte-to-white blood cell ratio and high monocyte-to-white blood cell ratio predict poor prognosis in gastric cancer</article-title><source>Oncotarget</source><year>2017</year><month>01</month><day>17</day><volume>8</volume><issue>3</issue><fpage>5281</fpage><lpage>5291</lpage><pub-id pub-id-type="doi">10.18632/oncotarget.14136</pub-id><pub-id pub-id-type="medline">28029656</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Suh</surname><given-names>B</given-names> </name><name name-style="western"><surname>Park</surname><given-names>S</given-names> </name><name name-style="western"><surname>Shin</surname><given-names>DW</given-names> </name><etal/></person-group><article-title>Low albumin-to-globulin ratio associated with cancer incidence and mortality in generally healthy adults</article-title><source>Ann Oncol</source><year>2014</year><month>11</month><volume>25</volume><issue>11</issue><fpage>2260</fpage><lpage>2266</lpage><pub-id pub-id-type="doi">10.1093/annonc/mdu274</pub-id><pub-id pub-id-type="medline">25057172</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thrift</surname><given-names>AP</given-names> </name><name name-style="western"><surname>Wenker</surname><given-names>TN</given-names> </name><name name-style="western"><surname>El-Serag</surname><given-names>HB</given-names> </name></person-group><article-title>Global burden of gastric cancer: epidemiological trends, risk factors, screening and prevention</article-title><source>Nat Rev Clin Oncol</source><year>2023</year><month>05</month><volume>20</volume><issue>5</issue><fpage>338</fpage><lpage>349</lpage><pub-id pub-id-type="doi">10.1038/s41571-023-00747-0</pub-id><pub-id pub-id-type="medline">36959359</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="web"><article-title>4forfull/IGC</article-title><source>GitHub</source><access-date>2026-07-30</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/4forfull/IGC">https://github.com/4forfull/IGC</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Grid search ranges and optimal parameter combinations for different machine learning algorithms.</p><media xlink:href="jmir_v28i1e94837_app1.docx" xlink:title="DOCX File, 18 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Baseline characteristics of the Guangdong Provincial People&#x2019;s Hospital (GDPH) and Shengli Oilfield Central Hospital (SOCH) cohort training sets after synthetic minority oversampling technique (SMOTE) augmentation.</p><media xlink:href="jmir_v28i1e94837_app2.docx" xlink:title="DOCX File, 21 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Flowchart of the study design.</p><media xlink:href="jmir_v28i1e94837_app3.docx" xlink:title="DOCX File, 1059 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Baseline characteristics of the external validation in Guangdong Provincial People&#x2019;s Hospital (GDPH) cohorts.</p><media xlink:href="jmir_v28i1e94837_app4.docx" xlink:title="DOCX File, 21 KB"/></supplementary-material><supplementary-material id="app5"><label>Multimedia Appendix 5</label><p>Baseline characteristics of the external validation in Shengli Oilfield Central Hospital (SOCH) cohorts.</p><media xlink:href="jmir_v28i1e94837_app5.docx" xlink:title="DOCX File, 22 KB"/></supplementary-material><supplementary-material id="app6"><label>Multimedia Appendix 6</label><p>Comparison of model performance across different correlation coefficients.</p><media xlink:href="jmir_v28i1e94837_app6.docx" xlink:title="DOCX File, 17 KB"/></supplementary-material><supplementary-material id="app7"><label>Multimedia Appendix 7</label><p>Age-stratified diagnostic performance of the model in the external validation cohort.</p><media xlink:href="jmir_v28i1e94837_app7.docx" xlink:title="DOCX File, 18 KB"/></supplementary-material></app-group></back></article>