<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e92921</article-id><article-id pub-id-type="doi">10.2196/92921</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Large Language Models for Heterogeneous Data Mining in Liver Disease: Framework Development and Retrospective Validation Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Haiping</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Li</surname><given-names>Xinming</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Fang</surname><given-names>Kechi</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ma</surname><given-names>Yinxue</given-names></name><degrees>BSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Li</surname><given-names>Lijuan</given-names></name><degrees>MLT</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yan</surname><given-names>Huiping</given-names></name><degrees>MM</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Liu</surname><given-names>Yanmin</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yu</surname><given-names>Yanhua</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Wang</surname><given-names>Jing</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref></contrib></contrib-group><aff id="aff1"><institution>Clinical Laboratory Center, Beijing Youan Hospital, Capital Medical University</institution><addr-line>Beijing</addr-line><country>P R China</country></aff><aff id="aff2"><institution>Clinical Research Center for Autoimmune Liver Disease, Beijing Youan Hospital, Capital Medical University</institution><addr-line>Beijing</addr-line><country>P R China</country></aff><aff id="aff3"><institution>State Key Laboratory of Cognitive Science and Mental Health, Institute of Psychology, Chinese Academy of Sciences</institution><addr-line>No. 16, Lincui Road, Chaoyang District</addr-line><addr-line>Beijing</addr-line><country>P R China</country></aff><aff id="aff4"><institution>Department of Psychology, University of Chinese Academy of Sciences</institution><addr-line>Beijing</addr-line><country>P R China</country></aff><aff id="aff5"><institution>Second Department of Liver Disease Center, Beijing Youan Hospital, Capital Medical University</institution><addr-line>Beijing</addr-line><country>P R China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Shen</surname><given-names>Bairong</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Sun</surname><given-names>Chunbao</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Jing Wang, PhD, State Key Laboratory of Cognitive Science and Mental Health, Institute of Psychology, Chinese Academy of Sciences, No. 16, Lincui Road, Chaoyang District, Beijing, 100101, P R China, 86 10-64855841; <email>wangjing@psych.ac.cn</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>4</day><month>9</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e92921</elocation-id><history><date date-type="received"><day>06</day><month>02</month><year>2026</year></date><date date-type="rev-recd"><day>14</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>15</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Haiping Zhang, Xinming Li, Kechi Fang, Yinxue Ma, Lijuan Li, Huiping Yan, Yanmin Liu, Yanhua Yu, Jing Wang. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 4.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e92921"/><abstract><sec><title>Background</title><p>Differentiating among liver disease entities such as autoimmune liver disease (AILD), drug-induced liver injury (DILI), and chronic hepatitis B (CHB) remains clinically challenging due to overlapping clinical manifestations and nonspecific laboratory findings. Conventional machine learning (ML) approaches rely mainly on structured laboratory data, whereas free-text clinical reports and other heterogeneous electronic medical record data are often underused. Large language models (LLMs) may provide a strategy for encoding heterogeneous clinical information, yet their usefulness for liver disease classification remains insufficiently evaluated.</p></sec><sec><title>Objective</title><p>This study aimed to evaluate the usefulness of LLM-derived embeddings for clinical data mining in liver disease and to determine whether integrating these embeddings with laboratory variables improves classification across broad disease categories and closely related subtypes.</p></sec><sec sec-type="methods"><title>Methods</title><p>We retrospectively analyzed electronic medical record data from 7543 patients with nonoverlapping liver disease etiologies treated at Beijing Youan Hospital, Capital Medical University, between 2010 and 2025. Three LLMs (Qwen3, Huatuo-o1, and II-Medical) generated semantic embeddings from standardized clinical text, combining free-text examination reports, and structured clinical observations. Performance was assessed in a 3-class etiological task (AILD, DILI, and CHB) and a 4-class task further subclassifying AILD into autoimmune hepatitis and primary biliary cholangitis. We compared embedding-only models, LLM-integrated ML models, and an ML-only baseline using the same structured variable set and preprocessing pipeline, with lightweight natural language processing encoders and zero-shot LLM reasoning as additional comparators. Models were developed using 5-fold cross-validation and evaluated on an internal holdout set using accuracy, macroaveraged precision, recall, and <italic>F</italic><sub>1</sub>-score.</p></sec><sec sec-type="results"><title>Results</title><p>In the 3-class task, the LLM-integrated ML models achieved macro <italic>F</italic><sub>1</sub>-scores of 0.835&#x2010;0.837, compared with 0.791 for the ML-only baseline, with corresponding accuracies of 0.925&#x2010;0.929 versus 0.893. In the 4-class task, the LLM-integrated ML models achieved macro <italic>F</italic><sub>1</sub>-scores of 0.717&#x2010;0.734, compared with 0.665 for the ML-only baseline, with corresponding accuracies of 0.920&#x2010;0.922 versus 0.874. A temporal split sensitivity analysis using cases from 2010 to 2019 for training and cases from 2020 to 2025 for testing showed that the relative advantage of LLM-integrated ML models over the ML-only baseline was preserved. Direct zero-shot LLM reasoning and lightweight natural language processing encoders performed below the embedding-based integrated models.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>In this single-center retrospective cohort of patients with clear-cut, nonoverlapping liver disease etiologies, LLM-derived embeddings provided complementary information to structured laboratory variables for multiclass liver disease classification. The integrated framework showed improved internal validation performance compared with the ML-only model, particularly for non-CHB categories and fine-grained subtype discrimination. Because patients with overlapping liver disease etiologies were excluded, the reported performance may overestimate diagnostic accuracy in broader real-world clinical settings where overlapping syndromes are common. Multicenter external validation and prospective evaluation in more heterogeneous patient populations are needed before clinical implementation.</p></sec></abstract><kwd-group><kwd>large language models</kwd><kwd>liver disease</kwd><kwd>clinical data mining</kwd><kwd>natural language processing</kwd><kwd>electronic medical records</kwd><kwd>LLMs</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Liver diseases span a wide spectrum of chronic conditions with diverse etiologies, and accurate diagnosis is crucial for appropriate treatment and improved prognosis [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. Autoimmune hepatitis (AIH), primary biliary cholangitis (PBC), drug-induced liver injury (DILI), and chronic hepatitis B (CHB) differ significantly in their pathogenesis and clinical management [<xref ref-type="bibr" rid="ref3">3</xref>-<xref ref-type="bibr" rid="ref6">6</xref>]. However, these conditions often exhibit overlapping symptoms and laboratory indicators, posing a significant challenge for differential diagnosis in routine clinical settings [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. Therefore, developing automated frameworks to assist in identifying liver disease subtypes is of substantial clinical importance [<xref ref-type="bibr" rid="ref9">9</xref>-<xref ref-type="bibr" rid="ref12">12</xref>].</p><p>The advent of modern hospital information systems has provided access to vast amounts of structured laboratory results and unstructured clinical narratives. However, these data are highly heterogeneous. Critical diagnostic information is often embedded in imaging and narrative reports, which are difficult to quantify using traditional statistical methods [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref14">14</xref>]. Furthermore, conventional machine learning (ML) pipelines rely heavily on manual feature engineering [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>], such as outlier handling and variable selection [<xref ref-type="bibr" rid="ref17">17</xref>], which are time-consuming and may fail to fully capture latent patterns within multisource clinical data.</p><p>Large language models (LLMs) have recently demonstrated exceptional capabilities in medical text understanding [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>]. LLMs can act as general-purpose feature extractors, transforming unstructured clinical text into dense semantic embeddings. This facilitates the integration of narrative information with structured data and reduces the need for manual preprocessing [<xref ref-type="bibr" rid="ref20">20</xref>]. Despite their potential, the use of LLM-based embeddings as feature encoding tools for downstream predictive modeling remains underexplored in liver disease research [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref22">22</xref>].</p><p>In this study, we developed an LLM-integrated framework for liver disease classification using heterogeneous electronic medical record (EMR) data. We hypothesized that semantic embeddings derived from free-text clinical reports and standardized clinical observations could capture diagnostic information not fully represented by structured laboratory variables alone. To test this hypothesis, we analyzed a real-world retrospective cohort of 7543 patients with liver disease treated between 2010 and 2025. We evaluated whether LLM-derived embeddings, alone or integrated with clinical laboratory variables, could improve classification across 2 diagnostic settings: broad etiological classification of autoimmune liver disease (AILD), DILI, and CHB, and fine-grained discrimination among AIH, PBC, DILI, and CHB. We further compared the integrated framework with ML-only models and exploratory direct zero-shot LLM reasoning.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>The overall study design is illustrated in <xref ref-type="fig" rid="figure1">Figure 1</xref>. This was a retrospective framework development and internal validation study based on EMR data from a single tertiary liver disease center. Eligible patients were identified from the EMR system, and structured laboratory indicators and unstructured clinical narratives were extracted. LLM-based semantic embeddings were generated from standardized clinical text inputs and integrated with clinical laboratory variables. Classification models were then developed and evaluated against ML-only pipelines and direct LLM inference.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Overview of the study design and modeling framework. Electronic medical record data, including free-text clinical reports and structured laboratory variables, were extracted and converted into standardized patient-level inputs. Three LLM encoders generated semantic embeddings, which were integrated with clinical laboratory variables for downstream ML classification. Model development was performed using stratified 5-fold cross-validation, followed by final evaluation on an internal holdout validation set. A direct zero-shot LLM reasoning pathway was assessed in parallel for comparison. The final classification framework was evaluated across 4 liver disease categories (autoimmune hepatitis, primary biliary cholangitis, drug-induced liver injury, and chronic hepatitis B). ALT: alanine aminotransferase; AST: aspartate aminotransferase; CT: computed tomographic; LLM: large language model; ML: machine learning; PCA: principal component analysis.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92921_fig01.png"/></fig></sec><sec id="s2-2"><title>Data Collection and Dataset Construction</title><p>We analyzed patient data retrieved from the EMR system of Beijing Youan Hospital, Capital Medical University. The initial screening included 162,159 patients seen between January 2010 and June 2025. The EMR contained structured laboratory test results and unstructured clinical narratives generated during routine clinical care, including imaging, endoscopy, and pathology-related reports, if available.</p><p>The study focused on patients diagnosed with CHB, DILI, and AILD. Within the AILD spectrum, AIH and PBC were included as the major subtypes. Patients with rarer autoimmune hepatobiliary diseases, including primary sclerosing cholangitis and IgG4-related sclerosing cholangitis, were excluded from the final analytical cohort because their limited sample sizes precluded robust statistical modeling.</p></sec><sec id="s2-3"><title>Inclusion and Exclusion Criteria</title><p>Inclusion criteria were (1) age &#x2265;18 years, (2) at least 2 independent laboratory records supporting the same diagnosis, and (3) availability of medication and treatment records. Diagnostic labels were assigned according to established international guidelines supplemented by Chinese national guidelines. Specifically, AIH was defined according to the International Autoimmune Hepatitis Group simplified criteria [<xref ref-type="bibr" rid="ref23">23</xref>], PBC according to the European Association for the Study of the Liver and Chinese Society of Hepatology guidelines [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>], DILI according to the Roussel Uclaf Causality Assessment Method [<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref27">27</xref>], and CHB according to the American Association for the Study of Liver Diseases along with the 2022 guidelines jointly issued by the Chinese Society of Hepatology and the Chinese Society of Infectious Diseases [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>].</p><p>Because the enrollment period spanned from 2010 to 2025, all historical cases were retrospectively readjudicated by a senior hepatologist using archived clinical, serological, laboratory, imaging, and treatment data. Cases that no longer met the current criteria or lacked sufficient evidence for diagnostic confirmation were excluded to ensure label consistency across the 15-year study window.</p><p>Exclusion criteria were (1) coexistence of multiple competing liver etiologies, such as AIH-PBC overlap syndrome or CHB combined with DILI; (2) major comorbidities likely to independently affect liver biochemistry, lipid metabolism, or imaging findings, including malignant tumors and diabetes mellitus; (3) alcohol-related liver disease; (4) duplicate records for which only the first eligible encounter with a confirmed diagnosis was retained; and (5) patient-level missingness exceeding 40% across candidate clinical variables. These criteria were applied to reduce etiological ambiguity and ensure diagnostic certainty in this initial multiclass classification task involving nonoverlapping target etiologies. Patients with common metabolic comorbidities other than diabetes mellitus, including hypertension, dyslipidemia, obesity, and metabolic dysfunction&#x2013;associated steatotic liver disease with no diabetes, were retained in the cohort to preserve generalizability to typical clinical populations.</p></sec><sec id="s2-4"><title>Dataset Partitioning</title><p>After applying the inclusion and exclusion criteria, the final cohort comprised 7543 patients across 3 disease categories, AILD (n=702), DILI (n=992), and CHB (n=5849). The AILD group consisted of AIH (n=256) and PBC (n=446). The cohort was stratified by the 4-class labels (AIH, PBC, DILI, and CHB) and randomly split into training (80%) and internal holdout validation (20%) sets. This stratification ensured class proportion preservation for both the 3-class task (where AIH and PBC are merged into AILD) and the 4-class task. Within the training set, we performed 5-fold cross-validation for dimensionality reduction and hyperparameter tuning [<xref ref-type="bibr" rid="ref30">30</xref>]. The internal holdout validation set was kept separate from all model development procedures and was used only for final performance evaluation [<xref ref-type="bibr" rid="ref31">31</xref>].</p></sec><sec id="s2-5"><title>Ethical Considerations</title><p>The study was approved by the Ethics Committee of Beijing Youan Hospital, Capital Medical University under protocol LL-2025&#x2010;029-K. The approved retrospective data use protocol covered the extraction and analysis of deidentified EMR data collected between January 2010 and June 2025. The requirement for informed consent was waived by the ethics committee. Data privacy was protected by deidentifying all records before analysis. All LLM encoding and reasoning tasks were performed locally to minimize the risk of information leakage.</p></sec><sec id="s2-6"><title>Construction of LLM Embedding and Integrated Features</title><p>Three 8B-parameter LLMs were selected as clinical data encoders, including Huatuo-o1 [<xref ref-type="bibr" rid="ref32">32</xref>], II-Medical [<xref ref-type="bibr" rid="ref33">33</xref>], and Qwen3 [<xref ref-type="bibr" rid="ref34">34</xref>]. Huatuo-o1 and II-Medical are domain-specific medical LLMs, while Qwen3 is a general-purpose LLM. Exact repository identifiers and checkpoint versions for local deployment of each model are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> to support computational reproducibility. To determine whether large medical LLMs provided incremental value over lighter text encoders, we also evaluated 2 lightweight natural language processing encoders, Doc2Vec [<xref ref-type="bibr" rid="ref35">35</xref>] and BGE-M3 [<xref ref-type="bibr" rid="ref36">36</xref>].</p><p>For each patient, structured and unstructured EMR contents were consolidated into a standardized modular text format. Examination items were organized as &#x201C;&#x003C;examination item&#x003E; : &#x003C;content&#x003E;&#x201D; and grouped into clinical modules, including demographics, laboratory tests, imaging reports, endoscopy reports, and pathology-related descriptions when available. To reduce potential label leakage before LLM encoding, explicit diagnostic labels, target disease names, discharge diagnoses, prescribed medication names, and other postdiagnostic information were removed or masked using predefined filtering rules [<xref ref-type="bibr" rid="ref37">37</xref>]. These rules were applied to both structured medication fields and unstructured free-text narratives to identify and mask medication names and disease-specific treatment keywords, including highly diagnosis-informative terms such as specific antiviral agents and ursodeoxycholic acid. The complete list of high-risk keywords used for filtering is provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. A processed example is shown in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p><p>For each patient, dense embeddings of dimension 4096 were extracted by mean-pooling the token-level hidden states of the final transformer layer of each LLM, followed by L2 normalization, in a zero-shot setting, without any task-specific fine-tuning. To balance computational efficiency with information retention, principal component analysis (PCA) was applied to project the embeddings into lower-dimensional representations. Three target dimensions, 32, 64, and 128, were systematically evaluated. PCA fitting was performed only within the training data during cross-validation to prevent information leakage. The resulting LLM embeddings were then integrated with clinical laboratory variables to construct LLM-integrated feature sets. This integration strategy preserved interpretable clinical measurements while adding latent semantic representations derived from unstructured clinical narratives.</p></sec><sec id="s2-7"><title>Prompt Engineering and Direct LLM Reasoning</title><p>To explore the direct clinical reasoning capability of LLMs, we implemented a structured prompt engineering workflow. Models were prompted via a structured template that specified the role definition, task instructions, input constraints, and standardized output formats for disease classification and medication recommendation [<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>]. The prompt template is shown in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>. All reasoning tasks were conducted in a zero-shot setting without task-specific fine-tuning [<xref ref-type="bibr" rid="ref40">40</xref>].</p><p>To improve reproducibility, deterministic generation was used whenever supported by the model. The temperature was set to 0, and sampling was disabled by setting do_sample to false. Detailed inference hyperparameters, including maximum output length and decoding settings for each model, are provided in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>. All LLM inference, including embedding extraction and zero-shot reasoning, was performed locally on a workstation equipped with 8 NVIDIA GeForce RTX 2080 Ti graphics processing units with 11 GB of video random access memory each.</p><p>Direct LLM diagnostic performance was evaluated by comparing the model-predicted disease category, parsed from the structured output field specified in the prompt template, with the reference diagnosis. Outputs that failed to match any of the predefined categories were classified as incorrect predictions.</p><p>Medication recommendation performance was assessed as an exploratory end point. Prescription records served as the reference list for comparison. To prevent label leakage, all medication-related content was removed from the input before it was supplied to the LLMs, and the models were prompted to generate medication recommendations from the remaining clinical narratives and laboratory data.</p><p>A medication match was defined as an exact or synonym-normalized match between a recommended drug class and a recorded medication class [<xref ref-type="bibr" rid="ref41">41</xref>]. For each patient, medication precision was defined as the number of correctly recommended drug classes divided by the total number of recommended drug classes, and medication recall was defined as the number of correctly recommended drug classes divided by the total number of prescribed drug classes. Patient-level macroaveraged precision, recall, and <italic>F</italic><sub>1</sub>-score were averaged across patients and reported as the primary medication recommendation metrics. When no medication was recommended, precision and <italic>F</italic><sub>1</sub>-score were set to 0 for that patient. The any-match rate was retained only as a secondary descriptive indicator. Because actual prescriptions may reflect disease severity, contraindications, physician preference, drug availability, and temporal changes in treatment strategy, medication matching was interpreted as an exploratory indicator of clinical plausibility rather than a measure of treatment correctness.</p></sec><sec id="s2-8"><title>ML-Only Modeling</title><p>For comparative analysis, an ML-only pipeline was developed using structured clinical laboratory variables without LLM-derived embeddings [<xref ref-type="bibr" rid="ref42">42</xref>]. To ensure a fair comparison with the LLM-integrated ML models, the ML-only model and the LLM-integrated ML models used the same structured clinical variable set and the same preprocessing pipeline. The only difference was that the LLM-integrated models additionally incorporated LLM-derived semantic embeddings.</p><p>For continuous variables, only those with at least 80% of nonmissing values (ie, &#x003C;20% missingness) were retained, and missing values were imputed using the k-nearest neighbors algorithm, which estimates missing values based on intersample similarity among clinical variables. For categorical variables, categorical missingness itself may carry clinical information (eg, a test not being ordered); missing values were therefore encoded as a separate missing category rather than imputed to retain potentially informative missingness patterns in real-world EMRs. This missingness-handling scheme was applied specifically to the variable set used for model development and differs from the descriptive statistics reported, which summarize the originally observed (nonimputed) values.</p><p>All retained clinical variables were used directly for modeling without additional feature selection or dimensionality reduction. To prevent information leakage, imputation and scaling were fitted only on the training subset within each cross-validation fold and were then applied to the corresponding validation subset [<xref ref-type="bibr" rid="ref43">43</xref>]. After hyperparameter selection, the complete preprocessing and modeling pipeline was refitted on the full development set and applied once to the internal holdout validation set for final performance evaluation.</p></sec><sec id="s2-9"><title>Model Training and Evaluation</title><p>Two classification tasks were implemented. The 3-class task classified patients as AILD, DILI, or CHB. The 4-class task classified patients as AIH, PBC, DILI, or CHB to evaluate fine-grained discrimination within AILD. Models were developed within the development set using stratified 5-fold cross-validation. Five downstream ML algorithms were evaluated in parallel, including random forest [<xref ref-type="bibr" rid="ref44">44</xref>], multilayer perceptron (MLP) [<xref ref-type="bibr" rid="ref45">45</xref>], logistic regression (LR) [<xref ref-type="bibr" rid="ref46">46</xref>], Extreme Gradient Boosting (XGBoost) [<xref ref-type="bibr" rid="ref47">47</xref>], and support vector machine (SVM) [<xref ref-type="bibr" rid="ref48">48</xref>]. For each candidate hyperparameter configuration, model training and validation were performed across all 5-folds, and the mean validation macro <italic>F</italic><sub>1</sub>-score was computed. The configuration with the highest mean validation macro <italic>F</italic><sub>1</sub>-score was selected for final deployment. The hyperparameter search space and fixed training parameters are provided in <xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref>.</p><p>Class imbalance was addressed during model training. For random forest, LR, and SVM, class_weight = 'balanced' was used to assign weights inversely proportional to class frequencies. For MLP and XGBoost, equivalent sample-weighting strategies were applied during training. These strategies were used only within the training folds [<xref ref-type="bibr" rid="ref49">49</xref>].</p><p>All preprocessing steps were incorporated into fold-specific training pipelines. For models requiring feature scaling, including MLP, LR, and SVM, <italic>z</italic> score normalization was fitted only on the training subset and then applied to the validation subset. PCA for LLM embeddings followed the same rule and was fitted only on the training subset within each fold. After cross-validation, the selected preprocessing steps and hyperparameter configuration were refitted on the full development set and evaluated once on the internal holdout validation set.</p><p>Model performance on the internal holdout validation set was assessed using accuracy, macro precision, macro recall, macro <italic>F</italic><sub>1</sub>-score, and macroaveraged area under the receiver operating characteristic curve. Macroaveraged metrics were used to assign equal weight to each disease class and reduce bias caused by the predominance of CHB cases. All performance metrics are reported with 95% CIs computed by stratified bootstrap resampling with 1000 iterations on the internal holdout validation set, preserving class proportions across iterations. The ML algorithm achieving the highest cross-validation performance within the training set was selected as the representative model and subsequently evaluated on the internal holdout validation set for cross-strategy comparison.</p><p>As a sensitivity analysis, we also performed temporal internal validation using cases from 2010 to 2019 for training and cases from 2020 to 2025 for testing. Hyperparameter tuning was conducted only within the 2010&#x2010;2019 training set using 5-fold cross-validation, and the 2020&#x2010;2025 temporal test set was evaluated once.</p></sec><sec id="s2-10"><title>Statistical Analysis</title><p>Normality of continuous variables was assessed once in the pooled cohort using D&#x2019;Agostino and Pearson omnibus test. Variables with fewer than 20 available observations were treated as non&#x2013;normally distributed by default, given insufficient power to assess normality. Regardless of distribution, continuous variables were summarized as median (IQR) for the overall cohort and for each of the 4 diagnostic groups (AIH, PBC, DILI, and CHB), together with the number of patients with an available result in that specific column (N); group comparisons used 1-way ANOVA for variables identified as normally distributed in the pooled cohort and the Kruskal-Wallis test otherwise, and were not reported (shown as "&#x2014;") when any of the 4 groups had fewer than 2 valid observations. Categorical variables were summarized as n/N (%) for the overall cohort and for each diagnostic group, where N is the number of patients with an available test result in that specific column rather than the corresponding column&#x2019;s full sample size, since not all laboratory and serological tests were obtained in every patient; group comparisons used the chi-square test and were not reported when any expected cell frequency was below 5. For each variable, the number and percentage of patients without an available result in the full cohort (missing, n [%]) were additionally reported, using the full cohort size as the denominator. All statistical tests were 2-sided, and <italic>P</italic>&#x003C;.05 was considered statistically significant. Statistical analyses were performed in Python (version 3.8; Python Software Foundation) using SciPy (version 1.7.3; SciPy Developers).</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Patients&#x2019; Clinical Characteristics</title><p>Baseline demographic and key clinical laboratory characteristics of the study cohort are summarized in <xref ref-type="table" rid="table1">Table 1</xref>. The internal holdout validation set exhibited distributions broadly comparable with those of the overall cohort, supporting its use for internal validation. Among the 7543 included patients, demographic features, biochemical profiles, viral hepatitis markers, autoimmune serological markers, and urinalysis findings varied across the 4 disease groups.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Baseline demographic and clinical laboratory characteristics of the study cohort by disease group<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Characteristic</td><td align="left" valign="bottom">Missing, n (%)</td><td align="left" valign="bottom">Total (N=7543)</td><td align="left" valign="bottom">AIH<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> (n=256)</td><td align="left" valign="bottom">PBC<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup> (n=446)</td><td align="left" valign="bottom">DILI<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup> (n=992)</td><td align="left" valign="bottom">CHB<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup> (n=5849)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Age (years), median (IQR)</td><td align="left" valign="top">0 (0.0)</td><td align="left" valign="top">44.0 (34.0-55.0), N=7543</td><td align="left" valign="top">54.0 (45.0-62.0), n=256</td><td align="left" valign="top">58.0 (50.0-65.0), n=446</td><td align="left" valign="top">50.0 (38.0-60.0), n=992</td><td align="left" valign="top">42.0 (32.0-52.0), n=5849</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Male, n/N (%)</td><td align="left" valign="top">0 (0.0)</td><td align="left" valign="top">3924/7543 (52.0)</td><td align="left" valign="top">35/256 (13.7)</td><td align="left" valign="top">64/446 (14.3)</td><td align="left" valign="top">347/992 (35.0)</td><td align="left" valign="top">3478/5849 (59.5)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">HBV<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup> DNA (+), n/N (%)</td><td align="left" valign="top">4792 (63.5)</td><td align="left" valign="top">1677/2751 (61.0)</td><td align="left" valign="top">0/10 (0.0)</td><td align="left" valign="top">1/18 (5.6)</td><td align="left" valign="top">1/57 (1.8)</td><td align="left" valign="top">1675/2666 (62.8)</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table1fn7">g</xref></sup></td></tr><tr><td align="left" valign="top">HBsAg<sup><xref ref-type="table-fn" rid="table1fn8">h</xref></sup> (+), n/N (%)</td><td align="left" valign="top">4995 (66.2)</td><td align="left" valign="top">1739/2548 (68.2)</td><td align="left" valign="top">1/94 (1.1)</td><td align="left" valign="top">2/135 (1.5)</td><td align="left" valign="top">6/436 (1.4)</td><td align="left" valign="top">1730/1883 (91.9)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Anti-HBs<sup><xref ref-type="table-fn" rid="table1fn9">i</xref></sup> (+), n/N (%)</td><td align="left" valign="top">1532 (20.3)</td><td align="left" valign="top">906/6011 (15.1)</td><td align="left" valign="top">55/123 (44.7)</td><td align="left" valign="top">93/204 (45.6)</td><td align="left" valign="top">257/579 (44.4)</td><td align="left" valign="top">501/5105 (9.8)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">HBeAg<sup><xref ref-type="table-fn" rid="table1fn10">j</xref></sup> (+), n/N (%)</td><td align="left" valign="top">1535 (20.3)</td><td align="left" valign="top">2059/6008 (34.3)</td><td align="left" valign="top">0/122 (0.0)</td><td align="left" valign="top">1/203 (0.5)</td><td align="left" valign="top">2/579 (0.3)</td><td align="left" valign="top">2056/5104 (40.3)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Anti-HBc<sup><xref ref-type="table-fn" rid="table1fn11">k</xref></sup> (+), n/N (%)</td><td align="left" valign="top">1535 (20.3)</td><td align="left" valign="top">5346/6008 (89.0)</td><td align="left" valign="top">46/122 (37.7)</td><td align="left" valign="top">95/203 (46.8)</td><td align="left" valign="top">177/579 (30.6)</td><td align="left" valign="top">5028/5104 (98.5)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">PreS1 Ag<sup><xref ref-type="table-fn" rid="table1fn12">l</xref></sup> (+), n/N (%)</td><td align="left" valign="top">4977 (66.0)</td><td align="left" valign="top">1894/2566 (73.8)</td><td align="left" valign="top">0/18 (0.0)</td><td align="left" valign="top">0/48 (0.0)</td><td align="left" valign="top">3/100 (3.0)</td><td align="left" valign="top">1891/2400 (78.8)</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table1fn7">g</xref></sup></td></tr><tr><td align="left" valign="top">Quantitative HBsAg titer (IU/mL), median (IQR)</td><td align="left" valign="top">1377 (18.3)</td><td align="left" valign="top">496.1 (130.0-3603.0), N=6166</td><td align="left" valign="top">0.0 (0.0-0.0), n=191</td><td align="left" valign="top">0.0 (0.0-0.0), n=334</td><td align="left" valign="top">0.0 (0.0-0.0), n=753</td><td align="left" valign="top">774.1 (130.0-4105.3), n=4888</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">ALT<sup><xref ref-type="table-fn" rid="table1fn13">m</xref></sup> (U/L), median (IQR)</td><td align="left" valign="top">4 (0.1)</td><td align="left" valign="top">40.0 (22.3-91.5), N=7539</td><td align="left" valign="top">50.0 (26.0-111.2), n=256</td><td align="left" valign="top">48.6 (24.1-83.0), n=444</td><td align="left" valign="top">106.3 (39.2-347.6), n=992</td><td align="left" valign="top">35.6 (21.0-73.0), n=5847</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">DBIL-to-TBIL<sup><xref ref-type="table-fn" rid="table1fn14">n</xref></sup>, median (IQR)</td><td align="left" valign="top">4 (0.1)</td><td align="left" valign="top">0.3 (0.2-0.4), N=7539</td><td align="left" valign="top">0.4 (0.3-0.6), n=256</td><td align="left" valign="top">0.4 (0.3-0.5), n=444</td><td align="left" valign="top">0.5 (0.3-0.7), n=992</td><td align="left" valign="top">0.3 (0.2-0.4), n=5847</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">GGT<sup><xref ref-type="table-fn" rid="table1fn15">o</xref></sup> (U/L), median (IQR)</td><td align="left" valign="top">94 (1.2)</td><td align="left" valign="top">38.0 (19.1-94.0), N=7449</td><td align="left" valign="top">85.3 (44.7-176.1), n=253</td><td align="left" valign="top">154.2 (54.0-330.1), n=439</td><td align="left" valign="top">110.5 (54.8-222.7), n=955</td><td align="left" valign="top">29.5 (16.7-62.0), n=5802</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">ALP<sup><xref ref-type="table-fn" rid="table1fn16">p</xref></sup> (U/L), median (IQR)</td><td align="left" valign="top">94 (1.2)</td><td align="left" valign="top">82.8 (63.5-115.0), N=7449</td><td align="left" valign="top">117.0 (85.0-177.0), n=253</td><td align="left" valign="top">169.0 (109.5-299.5), n=439</td><td align="left" valign="top">111.7 (82.2-159.1), n=955</td><td align="left" valign="top">76.0 (60.6-100.0), n=5802</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Globulin (g/L), median (IQR)</td><td align="left" valign="top">4 (0.1)</td><td align="left" valign="top">29.2 (26.2-32.7), N=7539</td><td align="left" valign="top">34.2 (30.3-39.9), n=256</td><td align="left" valign="top">35.1 (30.9-40.5), n=444</td><td align="left" valign="top">28.9 (25.7-32.4), n=992</td><td align="left" valign="top">28.8 (26.1-32.0), n=5847</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">eGFR<sup><xref ref-type="table-fn" rid="table1fn17">q</xref></sup> (mL/min/1.73m&#x00B2;), median (IQR)</td><td align="left" valign="top">566 (7.5)</td><td align="left" valign="top">111.7 (101.4-121.7), N=6977</td><td align="left" valign="top">105.6 (96.9-117.3), n=238</td><td align="left" valign="top">103.9 (95.0-111.6), n=429</td><td align="left" valign="top">109.1 (98.9-118.9), n=944</td><td align="left" valign="top">113.0 (103.1-122.7), n=5366</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">ASMA<sup><xref ref-type="table-fn" rid="table1fn18">r</xref></sup> (+), n/N (%)</td><td align="left" valign="top">4656 (61.7)</td><td align="left" valign="top">130/2887 (4.5)</td><td align="left" valign="top">26/217 (12.0)</td><td align="left" valign="top">4/372 (1.1)</td><td align="left" valign="top">31/674 (4.6)</td><td align="left" valign="top">69/1624 (4.2)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">ANA<sup><xref ref-type="table-fn" rid="table1fn19">s</xref></sup> (+), n/N (%)</td><td align="left" valign="top">4607 (61.1)</td><td align="left" valign="top">1969/2936 (67.1)</td><td align="left" valign="top">210/217 (96.8)</td><td align="left" valign="top">322/373 (86.3)</td><td align="left" valign="top">487/683 (71.3)</td><td align="left" valign="top">950/1663 (57.1)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">AMA<sup><xref ref-type="table-fn" rid="table1fn20">t</xref></sup> (+), n/N (%)</td><td align="left" valign="top">4670 (61.9)</td><td align="left" valign="top">411/2873 (14.3)</td><td align="left" valign="top">43/212 (20.3)</td><td align="left" valign="top">301/369 (81.6)</td><td align="left" valign="top">42/669 (6.3)</td><td align="left" valign="top">25/1623 (1.5)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Anticytoskeleton (+), n/N (%)</td><td align="left" valign="top">4656 (61.7)</td><td align="left" valign="top">80/2887 (2.8)</td><td align="left" valign="top">20/217 (9.2)</td><td align="left" valign="top">4/372 (1.1)</td><td align="left" valign="top">17/674 (2.5)</td><td align="left" valign="top">39/1624 (2.4)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Antiparietal cell (+), n/N (%)</td><td align="left" valign="top">4656 (61.7)</td><td align="left" valign="top">148/2887 (5.1)</td><td align="left" valign="top">13/217 (6.0)</td><td align="left" valign="top">7/372 (1.9)</td><td align="left" valign="top">45/674 (6.7)</td><td align="left" valign="top">83/1624 (5.1)</td><td align="left" valign="top">.008</td></tr><tr><td align="left" valign="top">Bilirubin (urine) (+), n/N (%)</td><td align="left" valign="top">4464 (59.2)</td><td align="left" valign="top">482/3079 (15.7)</td><td align="left" valign="top">24/85 (28.2)</td><td align="left" valign="top">30/173 (17.3)</td><td align="left" valign="top">193/519 (37.2)</td><td align="left" valign="top">235/2302 (10.2)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Nitrite (urine) (+), n/N (%)</td><td align="left" valign="top">4464 (59.2)</td><td align="left" valign="top">77/3079 (2.5)</td><td align="left" valign="top">3/85 (3.5)</td><td align="left" valign="top">13/173 (7.5)</td><td align="left" valign="top">12/519 (2.3)</td><td align="left" valign="top">49/2302 (2.1)</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table1fn7">g</xref></sup></td></tr><tr><td align="left" valign="top">Ketone (urine) (+), n/N (%)</td><td align="left" valign="top">4464 (59.2)</td><td align="left" valign="top">169/3079 (5.5)</td><td align="left" valign="top">3/85 (3.5)</td><td align="left" valign="top">4/173 (2.3)</td><td align="left" valign="top">24/519 (4.6)</td><td align="left" valign="top">138/2302 (6.0)</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table1fn7">g</xref></sup></td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup> Continuous variables are median (IQR), N; categorical variables are n/N (%). N is the number of patients with an available result in that specific column (total or subgroup) and may differ from the column header's nominal sample size. The &#x201C;Missing, n (%)&#x201D; column reflects overall cohort missingness only; subgroup-level completeness should be read from the N in each cell. Group comparisons used 1-way ANOVA or the Kruskal-Wallis test for continuous variables (per normality in the pooled cohort) and the chi-square test for categorical variables. </p></fn><fn id="table1fn2"><p><sup>b</sup>AIH: autoimmune hepatitis.</p></fn><fn id="table1fn3"><p><sup>c</sup>PBC: primary biliary cholangitis.</p></fn><fn id="table1fn4"><p><sup>d</sup>DILI: drug-induced liver injury.</p></fn><fn id="table1fn5"><p><sup>e</sup>CHB: chronic hepatitis B.</p></fn><fn id="table1fn6"><p><sup>f</sup>HBV: hepatitis B virus.</p></fn><fn id="table1fn7"><p><sup>g</sup><italic>P</italic> values are not reported (&#x201C;em dashes&#x201D;) when test assumptions were not met or results would be uninformative.</p></fn><fn id="table1fn8"><p><sup>h</sup>HBsAg: hepatitis B surface antigen.</p></fn><fn id="table1fn9"><p><sup>i</sup>Anti-HBs: antibody to hepatitis B surface antigen.</p></fn><fn id="table1fn10"><p><sup>j</sup>HBeAg: hepatitis B e antigen.</p></fn><fn id="table1fn11"><p><sup>k</sup>Anti-HBc: antibody to hepatitis B core antigen.</p></fn><fn id="table1fn12"><p><sup>l</sup>PreS1 Ag: pre-S1 antigen.</p></fn><fn id="table1fn13"><p><sup>m</sup>ALT: alanine aminotransferase. </p></fn><fn id="table1fn14"><p><sup>n</sup>DBIL-to-TBIL: direct bilirubin to total bilirubin ratio.</p></fn><fn id="table1fn15"><p><sup>o</sup>GGT: &#x03B3;-Glutamyl transferase.</p></fn><fn id="table1fn16"><p><sup>p</sup>ALP: alkaline phosphatase.</p></fn><fn id="table1fn17"><p><sup>q</sup>eGFR: estimated glomerular filtration rate.</p></fn><fn id="table1fn18"><p><sup>r</sup>ASMA: antismooth muscle antibody.</p></fn><fn id="table1fn19"><p><sup>s</sup>ANA: antinuclear antibody.</p></fn><fn id="table1fn20"><p><sup>t</sup>AMA: antimitochondrial antibody.</p></fn></table-wrap-foot></table-wrap><p>Patients with PBC and AIH were older and predominantly female, with median ages of 58.0 and 54.0 years and male proportions of 14.3% and 13.7%, respectively. In contrast, patients with CHB were younger and predominantly male, with a median age of 42.0 years and a male proportion of 59.5%. Biochemical profiles varied in patterns consistent with the corresponding disease categories. Cholestatic markers were highest in patients with PBC, with median &#x03B3;-glutamyl transferase and alkaline phosphatase values of 154.2 U/L and 169.0 U/L, respectively. Hepatocellular injury was most prominent in patients with DILI, with a median alanine aminotransferase of 106.3 U/L and the highest urinary bilirubin positivity rate (37.2%). Globulin levels were highest in patients with PBC (median 35.1 g/L), accompanied by lower estimated glomerular filtration rate values in both PBC and AIH compared with CHB, suggesting a degree of systemic involvement in autoimmune disease groups. Antimitochondrial antibody (AMA) positivity was frequent in PBC (81.6%), and antinuclear antibody (ANA) positivity was prevalent in both AIH and PBC (96.8% and 86.3%, respectively). Hepatitis B virus&#x2013;related markers were concentrated in CHB, including hepatitis B surface antigen positivity of 91.9% and hepatitis B e antigen positivity of 40.3%. These serological and biochemical patterns were consistent with established diagnostic patterns [<xref ref-type="bibr" rid="ref50">50</xref>-<xref ref-type="bibr" rid="ref54">54</xref>] and supported the validity of the cohort for downstream modeling. Complete statistical results for all variables are provided in <xref ref-type="supplementary-material" rid="app7">Multimedia Appendix 7</xref>.</p></sec><sec id="s3-2"><title>Feasibility of LLM-Based Embeddings in 3-Class Liver Disease Classification</title><p>To evaluate the discriminative potential of LLM-embedding models, we first conducted a 3-class task (AILD vs DILI vs CHB). As a reference, the ML-only model was developed using the full set of retained structured clinical laboratory variables without LLM-derived embeddings. This model achieved an accuracy of 0.893 (95% CI 0.877&#x2010;0.908) and a macro <italic>F</italic><sub>1</sub>-score of 0.791 (95% CI 0.761&#x2010;0.820) on the internal holdout validation set.</p><p>We then evaluated LLM embedding-only models using LLM-derived semantic embeddings as stand-alone features. These models were trained without expert-defined text feature engineering or manually selected semantic variables. With a PCA dimensionality of 64, the Qwen3 embedding-only model achieved the highest accuracy of 0.919 (95% CI 0.907&#x2010;0.932), exceeding the ML-only model. The Huatuo-o1 and II-Medical embedding-only models achieved comparable performance, indicating that LLM-derived embeddings alone retained clinically relevant discriminative information from heterogeneous EMR inputs (<xref ref-type="supplementary-material" rid="app8">Multimedia Appendix 8</xref>).</p><p>To contextualize the contribution of LLM-derived semantic representations, we compared their embedding-only models with 2 lightweight text representation methods. The 3 LLM embedding-only models outperformed Doc2Vec, which achieved an accuracy of 0.836 (95% CI 0.818&#x2010;0.854) and a macro <italic>F</italic><sub>1</sub>-score of 0.635 (95% CI 0.596&#x2010;0.672), and BGE-M3, which achieved an accuracy of 0.838 (95% CI 0.819&#x2010;0.856) and a macro <italic>F</italic><sub>1</sub>-score of 0.631 (95% CI 0.591&#x2010;0.669), on the internal holdout validation set (<xref ref-type="supplementary-material" rid="app9">Multimedia Appendix 9</xref>). These findings suggest that LLM-derived embeddings provided more discriminative semantic information than the lightweight text encoders evaluated in this study.</p><p>Taken together, embedding-only analysis showed that LLM-derived representations could encode clinically relevant information from standardized EMR inputs and provide competitive performance in broad etiological liver disease classification. These results support their use as high-dimensional feature representations for downstream ML classification.</p></sec><sec id="s3-3"><title>Complementary Value of Integrated LLM and Clinical Features</title><p>We next evaluated whether LLM-derived embeddings provided complementary information when combined with structured clinical laboratory variables. After feature integration, the LLM-integrated ML models achieved accuracies ranging from 0.925 to 0.929 on the internal holdout validation set, corresponding to improvements of 0.8-2.8 percentage points over the respective LLM embedding-only models (<xref ref-type="fig" rid="figure2">Figures 2A-2C</xref>). All 3 LLM-integrated ML models also outperformed the ML-only baseline, with accuracy increasing by 3.2-3.6 percentage points and macro <italic>F</italic><sub>1</sub>-score increasing by 4.4-4.6 percentage points (<xref ref-type="table" rid="table2">Table 2</xref>). Although the II-Medical&#x2013;integrated ML model achieved the highest accuracy of 0.929 (95% CI 0.916&#x2010;0.941), the 3 LLM-integrated models showed comparable overall performance, with overlapping 95% CIs across the reported metrics.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Three-class classification performance of large language model (LLM)&#x2013;embedding-only and LLM-integrated ML models on the internal holdout validation set. The figure compares model performance in the 3-class task distinguishing autoimmune liver disease, drug-induced liver injury, and chronic hepatitis B. LLM embedding-only models used LLM-derived semantic embeddings as stand-alone features, whereas LLM-integrated ML models combined LLM-derived embeddings with clinical laboratory variables. (A) Performance comparison based on Huatuo-o1 embeddings, (B) performance comparison based on II-Medical embeddings, and (C) performance comparison based on Qwen3 embeddings. Across the evaluated LLMs, the integrated models showed higher accuracy, macro precision, macro recall, and macro <italic>F</italic><sub>1</sub>-score than the corresponding embedding-only models. ML: machine learning.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92921_fig02.png"/></fig><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Three-class classification performance of ML<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>-only and LLM<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup>-integrated ML models on the internal holdout validation set<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Accuracy (95% CI)</td><td align="left" valign="bottom">Macro precision (95% CI)</td><td align="left" valign="bottom">Macro recall (95% CI)</td><td align="left" valign="bottom">Macro <italic>F</italic><sub>1</sub>-score (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">ML-only model</td><td align="left" valign="top">0.893 (0.877&#x2010;0.908)</td><td align="left" valign="top">0.766 (0.734&#x2010;0.797)</td><td align="left" valign="top">0.829 (0.799&#x2010;0.858)</td><td align="left" valign="top">0.791 (0.761&#x2010;0.820)</td></tr><tr><td align="left" valign="top">Huatuo-o1&#x2013;integrated ML model</td><td align="left" valign="top">0.925 (0.912&#x2010;0.938)</td><td align="left" valign="top">0.832 (0.800&#x2010;0.861)</td><td align="left" valign="top">0.839 (0.809&#x2010;0.869)</td><td align="left" valign="top">0.835 (0.805&#x2010;0.863)</td></tr><tr><td align="left" valign="top">II-Medical&#x2013;integrated ML model</td><td align="left" valign="top">0.929 (0.916&#x2010;0.941)</td><td align="left" valign="top">0.834 (0.803&#x2010;0.864)</td><td align="left" valign="top">0.843 (0.812&#x2010;0.874)</td><td align="left" valign="top">0.837 (0.808&#x2010;0.866)</td></tr><tr><td align="left" valign="top">Qwen3-integrated ML model</td><td align="left" valign="top">0.927 (0.913&#x2010;0.940)</td><td align="left" valign="top">0.838 (0.808&#x2010;0.869)</td><td align="left" valign="top">0.839 (0.809&#x2010;0.869)</td><td align="left" valign="top">0.837 (0.808&#x2010;0.866)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>ML: machine learning.</p></fn><fn id="table2fn2"><p><sup>b</sup>LLM: large language model.</p></fn><fn id="table2fn3"><p><sup>c</sup>The ML-only model used the full set of clinical laboratory variables without feature selection, and the LLM-integrated ML models combined LLM embeddings with the same full set of clinical variables. Macroaveraged metrics were computed by averaging each metric across classes, treating all classes equally regardless of prevalence.</p></fn></table-wrap-foot></table-wrap><p>Class-specific precision analysis showed that performance differences were more evident for AILD and DILI than for CHB (<xref ref-type="table" rid="table3">Table 3</xref>). AILD precision was 0.700 (95% CI 0.623&#x2010;0.777) in the ML-only model and ranged from 0.775 to 0.796 in the LLM-integrated ML models. DILI precision was 0.606 (95% CI 0.550&#x2010;0.661) in the ML-only model and ranged from 0.738 to 0.744 in the LLM-integrated ML models. In contrast, CHB precision remained high across all models, ranging from 0.977 to 0.991.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Class-specific precision of ML<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>-only and LLM<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup>-integrated ML models in the 3-class classification task on the internal holdout validation set.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">AILD<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup> precision (95% CI)</td><td align="left" valign="bottom">DILI<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup> precision (95% CI)</td><td align="left" valign="bottom">CHB<sup><xref ref-type="table-fn" rid="table3fn5">e</xref></sup> precision (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">ML-only model</td><td align="left" valign="top">0.700 (0.623&#x2010;0.777)</td><td align="left" valign="top">0.606 (0.550&#x2010;0.661)</td><td align="left" valign="top">0.991 (0.985&#x2010;0.996)</td></tr><tr><td align="left" valign="top">Huatuo-o1&#x2013;integrated ML model</td><td align="left" valign="top">0.775 (0.702&#x2010;0.846)</td><td align="left" valign="top">0.744 (0.684&#x2010;0.798)</td><td align="left" valign="top">0.977 (0.968&#x2010;0.985)</td></tr><tr><td align="left" valign="top">II-Medical&#x2013;integrated ML model</td><td align="left" valign="top">0.780 (0.706&#x2010;0.849)</td><td align="left" valign="top">0.738 (0.683&#x2010;0.798)</td><td align="left" valign="top">0.983 (0.974&#x2010;0.990)</td></tr><tr><td align="left" valign="top">Qwen3-integrated ML model</td><td align="left" valign="top">0.796 (0.722&#x2010;0.866)</td><td align="left" valign="top">0.741 (0.684&#x2010;0.799)</td><td align="left" valign="top">0.977 (0.967&#x2010;0.985)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>ML: machine learning.</p></fn><fn id="table3fn2"><p><sup>b</sup>LLM: large language model.</p></fn><fn id="table3fn3"><p><sup>c</sup>AILD: autoimmune liver disease.</p></fn><fn id="table3fn4"><p><sup>d</sup>DILI: drug-induced liver injury.</p></fn><fn id="table3fn5"><p><sup>e</sup>CHB: chronic hepatitis B.</p></fn></table-wrap-foot></table-wrap><p>We further assessed the computational cost of LLM-integrated ML models. Per-patient end-to-end inference time, including tokenization, LLM embedding extraction, dimensionality reduction, feature concatenation, and classifier prediction, was longer than that of the ML-only model but remained approximately 670&#x2010;685 ms per case in the local benchmark (<xref ref-type="supplementary-material" rid="app10">Multimedia Appendix 10</xref>). All models were deployed locally on graphics processing units, and the embedding-based pipeline used a single forward pass without autoregressive token decoding. This design explains the lower latency compared with API-based generative reasoning. These findings suggest that LLM-derived embeddings provide information complementary to structured clinical laboratory variables in the 3-class task. The contribution was most apparent for AILD and DILI, whereas CHB classification remained robust across modeling approaches.</p></sec><sec id="s3-4"><title>Performance in Refined 4-Class Classification Tasks</title><p>We next evaluated the feature integration strategy in a more refined 4-class classification task (AIH vs PBC vs DILI vs CHB) to assess whether LLM embeddings could further discriminate between clinically related AILD subtypes. The ML-only model achieved an accuracy of 0.874 (95% CI 0.857&#x2010;0.889) and a macro <italic>F</italic><sub>1</sub>-score of 0.665 (95% CI 0.626&#x2010;0.701) on the internal holdout validation set. In comparison, the 3 LLM-integrated ML models achieved accuracies ranging from 0.920 to 0.922, corresponding to improvements of 4.6-4.8 percentage points over the ML-only model (<xref ref-type="table" rid="table4">Table 4</xref>). Macro <italic>F</italic><sub>1</sub>-scores also increased, ranging from 0.717 to 0.734 across the LLM-integrated models. The Huatuo-o1&#x2013;integrated ML model achieved the highest macro <italic>F</italic><sub>1</sub>-score of 0.734 (95% CI 0.690&#x2010;0.774), while the 3 LLM-integrated ML models showed broadly comparable performance with overlapping 95% CIs across the reported metrics.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Four-class classification performance of ML<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup>-only and LLM<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup>-integrated ML models on the internal holdout validation set.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Accuracy (95% CI)</td><td align="left" valign="bottom">Macro precision (95% CI)</td><td align="left" valign="bottom">Macro recall (95% CI)</td><td align="left" valign="bottom">Macro <italic>F</italic><sub>1</sub>-score (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">ML-only model</td><td align="left" valign="top">0.874 (0.857&#x2010;0.889)</td><td align="left" valign="top">0.635 (0.599&#x2010;0.672)</td><td align="left" valign="top">0.707 (0.664&#x2010;0.750)</td><td align="left" valign="top">0.665 (0.626&#x2010;0.701)</td></tr><tr><td align="left" valign="top">Huatuo-o1&#x2013;integrated ML model</td><td align="left" valign="top">0.922 (0.907&#x2010;0.935)</td><td align="left" valign="top">0.750 (0.701&#x2010;0.800)</td><td align="left" valign="top">0.730 (0.691&#x2010;0.769)</td><td align="left" valign="top">0.734 (0.690&#x2010;0.774)</td></tr><tr><td align="left" valign="top">II-Medical&#x2013;integrated ML model</td><td align="left" valign="top">0.920 (0.907&#x2010;0.934)</td><td align="left" valign="top">0.735 (0.686&#x2010;0.787)</td><td align="left" valign="top">0.717 (0.680&#x2010;0.757)</td><td align="left" valign="top">0.717 (0.677&#x2010;0.759)</td></tr><tr><td align="left" valign="top">Qwen3-integrated ML model</td><td align="left" valign="top">0.921 (0.907&#x2010;0.934)</td><td align="left" valign="top">0.746 (0.695&#x2010;0.802)</td><td align="left" valign="top">0.725 (0.690&#x2010;0.765)</td><td align="left" valign="top">0.727 (0.685&#x2010;0.769)</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>ML: machine learning.</p></fn><fn id="table4fn2"><p><sup>b</sup>LLM: large language model.</p></fn></table-wrap-foot></table-wrap><p>The LLM-integrated ML models also showed more balanced discrimination across disease classes, with higher macroaveraged area under the receiver operating characteristic curves than the ML-only model in the 4-class task (<xref ref-type="supplementary-material" rid="app11">Multimedia Appendix 11</xref>). These findings suggest that LLM-derived embeddings provided complementary information for the refined 4-class task rather than improving performance only in the majority CHB category.</p><p>Feature importance analysis identified both conventional clinical variables and LLM-derived components among the most informative features. Autoantibody markers, including AMA and ANA, remained highly ranked, consistent with their established relevance in AILD classification [<xref ref-type="bibr" rid="ref55">55</xref>]. Several LLM-derived principal components, including PCA_2, PCA_18, and PCA_8, were also among the top-ranked features (<xref ref-type="fig" rid="figure3">Figure 3A</xref>). These components represent orthogonal dimensions of the compressed embedding space and should be interpreted as semantic representation features rather than directly interpretable clinical variables.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Feature importance of the large language model (LLM)&#x2013;integrated machine learning (ML) model and comparison with direct LLM reasoning. (A) Top 40 features ranked by importance in the Huatuo-o1&#x2013;integrated ML model. Conventional clinical variables, including viral hepatitis markers, autoantibodies, and liver function indicators, ranked among the most informative features. Several LLM-derived principal components, including PCA_2, PCA_18, and PCA_8, were also retained among the top-ranked features. (B) Performance comparison between the Huatuo-o1&#x2013;integrated ML model and direct zero-shot reasoning by the 3 LLMs. The integrated ML model showed substantially higher accuracy (0.922 vs 0.515-0.721), macro precision (0.750 vs 0.459-0.586), macro recall (0.730 vs 0.440-0.575), and macro <italic>F</italic><sub>1</sub>-score (0.734 vs 0.337-0.477) compared with direct LLM reasoning across all 3 models. AFP: alpha-fetoprotein; ALB-to-GLO: albumin to globulin ratio; ALP: alkaline phosphatase; ALT: alanine aminotransferase; AMA: antimitochondrial antibody; ANA: antinuclear antibody; Anti-CM: anticardiac muscle antibody; Anti-HBc: antibody to hepatitis B core antigen; Anti-HBs: antibody to hepatitis B surface antigen; ApoB: apolipoprotein B; ASMA: antismooth muscle antibody; AST: aspartate aminotransferase; DBIL-to-TBIL: direct bilirubin to total bilirubin ratio; eGFR: estimated glomerular filtration rate; HBeAg: hepatitis B e antigen; HBsAg: hepatitis B surface antigen; HBV: hepatitis B virus; HCV: hepatitis C virus; HDL-C: high-density lipoprotein cholesterol; NRBC: nucleated red blood cells; PCA: principal component analysis; PCT: plateletcrit; PLT: platelet count; PreS1 Ag: pre-S1 antigen.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e92921_fig03.png"/></fig><p>Together, these findings support the complementary value of LLM-derived representations when integrated with structured clinical variables for fine-grained liver disease classification.</p></sec><sec id="s3-5"><title>Temporal Split Sensitivity Analysis</title><p>To assess model stability under temporal evaluation, we performed a sensitivity analysis using cases from 2010 to 2019 for training and cases from 2020 to 2025 for testing. Hyperparameter tuning was performed only within the 2010&#x2010;2019 training set, and the 2020&#x2010;2025 temporal test set was evaluated once.</p><p>In the 3-class task, the LLM-integrated ML models maintained higher macro <italic>F</italic><sub>1</sub>-scores than the ML-only baseline under temporal evaluation, with macro <italic>F</italic><sub>1</sub>-scores ranging from 0.852 to 0.858 compared with 0.799 for the ML-only model. In the 4-class task, the same pattern was observed, with macro <italic>F</italic><sub>1</sub>-scores ranging from 0.743 to 0.752 for the LLM-integrated ML models compared with 0.702 for the ML-only model. These findings indicate that the relative advantage of LLM-derived embeddings over the ML-only baseline was preserved under temporal internal validation. Full temporal split results are provided in <xref ref-type="supplementary-material" rid="app12">Multimedia Appendix 12</xref>.</p></sec><sec id="s3-6"><title>Exploratory Assessment of Direct LLM Reasoning</title><p>Beyond the embedding-based integration framework, we examined whether LLMs could directly generate disease classification and medication recommendations through structured prompts. In direct diagnostic inference, Qwen3 achieved the highest performance among the 3 evaluated LLMs, with an accuracy of 0.721 and a macro <italic>F</italic><sub>1</sub>-score of 0.477. However, this performance remained lower than that of the LLM-integrated ML framework. For example, the Huatuo-o1&#x2013;integrated ML model achieved an accuracy of 0.922 and a macro <italic>F</italic><sub>1</sub>-score of 0.734 in the 4-class task (<xref ref-type="fig" rid="figure3">Figure 3B</xref>).</p><p>Medication recommendation was evaluated as an exploratory end point by comparing model-generated drug classes with recorded prescriptions after medication name normalization. Agreement with actual prescriptions varied across models. II-Medical achieved the highest any-match rate of 57.5%, followed by Qwen3 at 52.6% and Huatuo-o1 at 18.9% (<xref ref-type="supplementary-material" rid="app13">Multimedia Appendix 13</xref>). The same pattern was observed using list-level metrics. Patient-level macro precision was 0.542 for II-Medical, 0.493 for Qwen3, and 0.184 for Huatuo-o1. Patient-level macro <italic>F</italic><sub>1</sub>-score was 0.280 for II-Medical, 0.249 for Qwen3, and 0.088 for Huatuo-o1.</p><p>The low recommendation rate observed for Huatuo-o1 was partly related to its tendency to recommend observation or no specific treatment for a substantial proportion of patients with CHB. This may represent clinically appropriate judgment but results in low overlap with recorded prescriptions under the current evaluation framework.</p><p>Overall, direct zero-shot LLM reasoning showed limited stand-alone diagnostic and medication recommendation performance compared with the embedding-based integration framework. These findings support the use of LLMs primarily as feature encoders within a controlled ML pipeline rather than as stand-alone diagnostic or therapeutic decision systems.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>In this single-center retrospective study, we developed an LLM-integrated framework for multiclass liver disease classification using heterogeneous EMR data. The framework combined semantic embeddings derived from free-text clinical reports and standardized clinical observations with structured laboratory variables. On the internal holdout validation set, the integrated models achieved higher performance than the ML-only model in both the 3-class task and the more granular 4-class task. The improvement was most evident for non-CHB categories and for fine-grained classification involving AIH and PBC, 2 AILD subtypes with overlapping biochemical profiles but different clinical management pathways.</p><p>A central finding of this study is that LLM-derived embeddings may provide information complementary to conventional laboratory variables. Clinical laboratory markers remain essential for liver disease classification, particularly viral hepatitis markers and autoantibodies such as ANA and AMA. However, free-text examination reports and other clinical narratives may contain contextual information that is difficult to represent through predefined structured variables alone. By encoding these heterogeneous inputs into high-dimensional representations, the LLM-integrated framework provided a way to incorporate narrative information into downstream ML models without requiring extensive rule-based text structuring.</p><p>The analysis also supports the value of evaluating models with metrics appropriate for imbalanced multiclass data. Because CHB represented the largest disease group, overall accuracy alone could overemphasize performance in the majority class. We therefore reported macroaveraged precision, recall, and <italic>F</italic><sub>1</sub>-score, together with class-specific results. These analyses showed that the LLM-integrated models improved macrolevel performance and performed better in non-CHB categories than the ML-only model. This pattern suggests that LLM-derived embeddings contributed most to disease categories with greater clinical heterogeneity rather than simply improving recognition of the dominant CHB group.</p><p>Feature importance analyses provided additional support for the complementary role of LLM-derived features. Conventional markers, including viral hepatitis markers, autoantibodies, and liver function indicators, remained among the most informative features. Several LLM-derived principal components were also ranked among the top features, suggesting that the embedding space retained discriminative information not fully captured by structured variables. However, these principal components should be interpreted as compressed semantic representations rather than as directly interpretable clinical concepts.</p><p>We also evaluated zero-shot LLM reasoning for liver disease classification and medication recommendation. Although this analysis showed exploratory value, direct LLM reasoning was less stable and less accurate than the embedding-based ML framework, and results should be interpreted with caution given the known sensitivity of zero-shot performance to prompt phrasing and instruction-tuning differences across models. Medication recommendation overlap should also be interpreted cautiously because prescriptions depend on disease severity, contraindications, patient history, and physician judgment. These findings support the use of LLMs as feature encoders within a controlled modeling pipeline rather than stand-alone diagnostic or therapeutic decision systems.</p></sec><sec id="s4-2"><title>Comparison With Prior Work</title><p>Previous liver disease prediction studies have often relied on selected biochemical markers, serological indicators, or quantitative imaging features [<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref57">57</xref>]. These approaches remain clinically meaningful but may not fully capture information contained in free-text reports and other semistructured clinical records. Recent studies have begun to explore LLMs for clinical text understanding, information extraction, and diagnostic reasoning, but their role as embedding-based feature encoders for liver disease classification remains insufficiently studied [<xref ref-type="bibr" rid="ref58">58</xref>,<xref ref-type="bibr" rid="ref59">59</xref>].</p><p>Our study extends this work by evaluating LLM-derived embeddings in a real-world liver disease cohort and by comparing several modeling strategies, including ML-only models, LLM embedding-only models, LLM-integrated ML models, lightweight natural language processing baselines, and direct zero-shot LLM reasoning. This comparative design helps distinguish the contribution of semantic representation learning from that of downstream ML classifiers. The results suggest that LLM-derived embeddings can improve internal classification performance when integrated with structured clinical variables, especially in diagnostically challenging categories.</p></sec><sec id="s4-3"><title>Limitations</title><p>This study has several limitations. First, this was a single-center retrospective study, and primary evaluation used an internal holdout validation set rather than an external cohort. Although historical cases were retrospectively readjudicated using updated diagnostic criteria, multicenter external validation is still needed to assess generalizability across hospitals, laboratory systems, and documentation styles. A temporal internal validation using cases from 2010 to 2019 for training and cases from 2020 to 2025 for testing showed that the relative advantage of the LLM-integrated models over the ML-only baseline was preserved (<xref ref-type="supplementary-material" rid="app12">Multimedia Appendix 12</xref>). However, this analysis remains an internal validation and does not substitute for external validation across independent centers. Second, predefined exclusions were applied to reduce etiological ambiguity, including the exclusion of patients with multiple coexisting liver disease etiologies and diabetes mellitus. Therefore, whether the framework generalizes to patients with overlapping liver disease mechanisms, substantial metabolic comorbidities, or metabolic dysfunction&#x2013;associated steatotic liver disease as an independent diagnostic category requires further validation. Third, although missingness thresholds and sensitivity analyses were applied, residual bias from incomplete retrospective EMR data and imputation cannot be fully excluded. In particular, missingness in serological markers such as hepatitis B surface antigen, ANA, and AMA reflects selective test ordering rather than random unavailability. The missing-indicator code used for these variables may therefore partly capture this ordering pattern rather than a purely biological signal, and disentangling the 2 would require prospective or external validation. Fourth, although explicit diagnostic labels, prescribed medication names, and disease-specific treatment keywords were removed or masked before LLM encoding, narrative mentions of specific treatments embedded in free-text reports may not have been completely captured by the rule-based filtering process, representing a potential source of residual label leakage. Fifth, due to hardware constraints, only 8B-parameter LLMs were evaluated, and the impact of larger models or domain-specific fine-tuning on representation quality remains unknown. Sixth, PCA-compressed embedding features cannot be directly mapped to individual clinical concepts, limiting interpretability. Finally, the zero-shot reasoning and medication recommendation analyses were exploratory. Zero-shot performance is sensitive to prompt phrasing, and the use of a single fixed template may have favored certain models. Medication matching was based on prescription overlap and should not be interpreted as evidence of treatment appropriateness or clinical usefulness. Prospective workflow-based evaluation is needed before clinical implementation.</p></sec><sec id="s4-4"><title>Conclusions</title><p>In this single-center retrospective cohort, LLM-derived embeddings provided complementary representations to structured laboratory data for multiclass liver disease classification. The LLM-integrated framework showed improved internal validation performance compared with ML-only models, particularly for non-CHB categories and fine-grained subtype discrimination involving AIH and PBC. These findings support further evaluation of LLM-derived embeddings as part of structured clinical prediction pipelines, but multicenter external validation and prospective assessment are required before integration into clinical workflows.</p></sec><sec id="s4-5"><title>Future Work</title><p>Future work should evaluate this framework in multicenter cohorts with different laboratory systems, documentation styles, and disease distributions. Prospective studies are also needed to determine whether the model can improve diagnostic workflow, clinician agreement, or subsequent test ordering in real-world settings. Additional work should focus on improving the interpretability of LLM-derived embeddings and assessing whether similar approaches are useful in other clinical domains where free-text reports provide information complementary to structured laboratory data.</p></sec></sec></body><back><ack><p>In this study, generative AI models were used for the grammar check of selected technical terms and for language polishing of the manuscript.</p></ack><notes><sec><title>Funding</title><p>This research received no external funding.</p></sec><sec><title>Data Availability</title><p>The raw datasets generated and analyzed during this study are not publicly available because of proprietary rights and data protection policies. The analysis code has been deposited in a public GitHub repository [<xref ref-type="bibr" rid="ref60">60</xref>].</p></sec></notes><fn-group><fn fn-type="con"><p>HZ, XL, and KF contributed equally to this work as co-first authors. HY, YL, YY, and JW contributed equally to this work as co-corresponding authors. JW contributed to conceptualization, supervision, and writing &#x2013; review &#x0026; editing. HY contributed to conceptualization and Supervisionsupervision. YL participated in supervision and writing &#x2013; review &#x0026; editing. YY participated in supervision, resources, and writing &#x2013; review &#x0026; editing. HZ participated in investigation, data curation, and writing &#x2013; review &#x0026; editing. XL developed the methodology, formal analysis, and writing &#x2013; review &#x0026; editing. KF developed the methodology, visualization, and writing &#x2013; review &#x0026; editing. YM participated in writing &#x2013; review &#x0026; editing. LL participated in writing &#x2013; review &#x0026; editing. All authors approved the final submitted version.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AIH</term><def><p>autoimmune hepatitis</p></def></def-item><def-item><term id="abb2">AILD</term><def><p>autoimmune liver disease</p></def></def-item><def-item><term id="abb3">AMA</term><def><p>antimitochondrial antibody</p></def></def-item><def-item><term id="abb4">ANA</term><def><p>antinuclear antibody</p></def></def-item><def-item><term id="abb5">CHB</term><def><p>chronic hepatitis B</p></def></def-item><def-item><term id="abb6">DILI</term><def><p>drug-induced liver injury</p></def></def-item><def-item><term id="abb7">EMR</term><def><p>electronic medical record</p></def></def-item><def-item><term id="abb8">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb9">LR</term><def><p>logistic regression</p></def></def-item><def-item><term id="abb10">ML</term><def><p>machine learning</p></def></def-item><def-item><term id="abb11">MLP</term><def><p>multilayer perceptron</p></def></def-item><def-item><term id="abb12">PBC</term><def><p>primary biliary cholangitis</p></def></def-item><def-item><term id="abb13">PCA</term><def><p>principal component analysis</p></def></def-item><def-item><term id="abb14">SVM</term><def><p>support vector machine</p></def></def-item><def-item><term id="abb15">XGBoost</term><def><p>Extreme Gradient Boosting</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Manikat</surname><given-names>R</given-names> </name><name name-style="western"><surname>Ahmed</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>D</given-names> </name></person-group><article-title>Current epidemiology of chronic liver disease</article-title><source>Gastroenterol Rep (Oxf)</source><year>2024</year><volume>12</volume><fpage>goae069</fpage><pub-id pub-id-type="doi">10.1093/gastro/goae069</pub-id><pub-id pub-id-type="medline">38915345</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moon</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Singal</surname><given-names>AG</given-names> </name><name name-style="western"><surname>Tapper</surname><given-names>EB</given-names> </name></person-group><article-title>Contemporary epidemiology of chronic liver disease and cirrhosis</article-title><source>Clin Gastroenterol Hepatol</source><year>2020</year><month>11</month><volume>18</volume><issue>12</issue><fpage>2650</fpage><lpage>2666</lpage><pub-id pub-id-type="doi">10.1016/j.cgh.2019.07.060</pub-id><pub-id pub-id-type="medline">31401364</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sucher</surname><given-names>E</given-names> </name><name name-style="western"><surname>Sucher</surname><given-names>R</given-names> </name><name name-style="western"><surname>Gradistanac</surname><given-names>T</given-names> </name><name name-style="western"><surname>Brandacher</surname><given-names>G</given-names> </name><name name-style="western"><surname>Schneeberger</surname><given-names>S</given-names> </name><name name-style="western"><surname>Berg</surname><given-names>T</given-names> </name></person-group><article-title>Autoimmune hepatitis-immunologically triggered liver pathogenesis-diagnostic and therapeutic strategies</article-title><source>J Immunol Res</source><year>2019</year><volume>2019</volume><fpage>9437043</fpage><pub-id pub-id-type="doi">10.1155/2019/9437043</pub-id><pub-id pub-id-type="medline">31886312</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Curto</surname><given-names>A</given-names> </name><name name-style="western"><surname>Iamello</surname><given-names>RG</given-names> </name><name name-style="western"><surname>Lynch</surname><given-names>EN</given-names> </name><name name-style="western"><surname>Galli</surname><given-names>A</given-names> </name></person-group><article-title>Advancing the management of primary biliary cholangitis: from pathogenesis to emerging therapies</article-title><source>World J Clin Cases</source><year>2025</year><month>10</month><day>26</day><volume>13</volume><issue>30</issue><fpage>109028</fpage><pub-id pub-id-type="doi">10.12998/wjcc.v13.i30.109028</pub-id><pub-id pub-id-type="medline">41113084</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chalasani</surname><given-names>N</given-names> </name><name name-style="western"><surname>Fontana</surname><given-names>RJ</given-names> </name><name name-style="western"><surname>Bonkovsky</surname><given-names>HL</given-names> </name><etal/></person-group><article-title>Causes, clinical features, and outcomes from a prospective study of drug-induced liver injury in the United States</article-title><source>Gastroenterology</source><year>2008</year><month>12</month><volume>135</volume><issue>6</issue><fpage>1924</fpage><lpage>1934</lpage><pub-id pub-id-type="doi">10.1053/j.gastro.2008.09.011</pub-id><pub-id pub-id-type="medline">18955056</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mak</surname><given-names>LY</given-names> </name><name name-style="western"><surname>Seto</surname><given-names>WK</given-names> </name><name name-style="western"><surname>Yuen</surname><given-names>MF</given-names> </name></person-group><article-title>Novel antivirals in clinical development for chronic hepatitis B infection</article-title><source>Viruses</source><year>2021</year><month>06</month><day>18</day><volume>13</volume><issue>6</issue><fpage>1169</fpage><pub-id pub-id-type="doi">10.3390/v13061169</pub-id><pub-id pub-id-type="medline">34207458</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schroeder</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Matsukuma</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Medici</surname><given-names>V</given-names> </name></person-group><article-title>Wilson disease and the differential diagnosis of its hepatic manifestations: a narrative review of clinical, laboratory, and liver histological features</article-title><source>Ann Transl Med</source><year>2021</year><month>09</month><volume>9</volume><issue>17</issue><fpage>1394</fpage><pub-id pub-id-type="doi">10.21037/atm-21-2264</pub-id><pub-id pub-id-type="medline">34733946</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Corrigan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hirschfield</surname><given-names>GM</given-names> </name></person-group><article-title>Autoimmune liver disease: evaluating overlapping and cross-over presentations-a case-based discussion</article-title><source>Frontline Gastroenterol</source><year>2016</year><month>10</month><volume>7</volume><issue>4</issue><fpage>240</fpage><lpage>245</lpage><pub-id pub-id-type="doi">10.1136/flgastro-2016-100698</pub-id><pub-id pub-id-type="medline">28839864</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Listopad</surname><given-names>S</given-names> </name><name name-style="western"><surname>Magnan</surname><given-names>C</given-names> </name><name name-style="western"><surname>Asghar</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Differentiating between liver diseases by applying multiclass machine learning approaches to transcriptomics of liver tissue or blood-based samples</article-title><source>JHEP Rep</source><year>2022</year><month>10</month><volume>4</volume><issue>10</issue><fpage>100560</fpage><pub-id pub-id-type="doi">10.1016/j.jhepr.2022.100560</pub-id><pub-id pub-id-type="medline">36119721</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Shu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>S</given-names> </name><etal/></person-group><article-title>A machine learning-based model analysis for serum markers of liver fibrosis in chronic hepatitis B patients</article-title><source>Sci Rep</source><year>2024</year><month>05</month><day>27</day><volume>14</volume><issue>1</issue><fpage>12081</fpage><pub-id pub-id-type="doi">10.1038/s41598-024-63095-8</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ganie</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Dutta Pramanik</surname><given-names>PK</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>Z</given-names> </name></person-group><article-title>Improved liver disease prediction from clinical data through an evaluation of ensemble learning approaches</article-title><source>BMC Med Inform Decis Mak</source><year>2024</year><month>06</month><day>7</day><volume>24</volume><issue>1</issue><fpage>160</fpage><pub-id pub-id-type="doi">10.1186/s12911-024-02550-y</pub-id><pub-id pub-id-type="medline">38849815</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nishida</surname><given-names>N</given-names> </name><name name-style="western"><surname>Kudo</surname><given-names>M</given-names> </name></person-group><article-title>Artificial intelligence models for the diagnosis and management of liver diseases</article-title><source>Ultrasonography</source><year>2023</year><month>01</month><volume>42</volume><issue>1</issue><fpage>10</fpage><lpage>19</lpage><pub-id pub-id-type="doi">10.14366/usg.22110</pub-id><pub-id pub-id-type="medline">36443931</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xiong</surname><given-names>FX</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>XJ</given-names> </name><etal/></person-group><article-title>Machine learning-based models for advanced fibrosis in non-alcoholic steatohepatitis patients: a cohort study</article-title><source>World J Gastroenterol</source><year>2025</year><month>03</month><day>7</day><volume>31</volume><issue>9</issue><fpage>101383</fpage><pub-id pub-id-type="doi">10.3748/wjg.v31.i9.101383</pub-id><pub-id pub-id-type="medline">40061588</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Weiskopf</surname><given-names>NG</given-names> </name><name name-style="western"><surname>Weng</surname><given-names>C</given-names> </name></person-group><article-title>Methods and dimensions of electronic health record data quality assessment: enabling reuse for clinical research</article-title><source>J Am Med Inform Assoc</source><year>2013</year><month>01</month><day>1</day><volume>20</volume><issue>1</issue><fpage>144</fpage><lpage>151</lpage><pub-id pub-id-type="doi">10.1136/amiajnl-2011-000681</pub-id><pub-id pub-id-type="medline">22733976</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Roe</surname><given-names>KD</given-names> </name><name name-style="western"><surname>Jawa</surname><given-names>V</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Feature engineering with clinical expert knowledge: a case study assessment of machine learning model complexity and performance</article-title><source>PLoS One</source><year>2020</year><volume>15</volume><issue>4</issue><fpage>e0231300</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0231300</pub-id><pub-id pub-id-type="medline">32324754</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Honeyford</surname><given-names>K</given-names> </name><name name-style="western"><surname>Expert</surname><given-names>P</given-names> </name><name name-style="western"><surname>Mendelsohn</surname><given-names>EE</given-names> </name><etal/></person-group><article-title>Challenges and recommendations for high quality research using electronic health records</article-title><source>Front Digit Health</source><year>2022</year><volume>4</volume><fpage>940330</fpage><pub-id pub-id-type="doi">10.3389/fdgth.2022.940330</pub-id><pub-id pub-id-type="medline">36060540</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hassler</surname><given-names>AP</given-names> </name><name name-style="western"><surname>Menasalvas</surname><given-names>E</given-names> </name><name name-style="western"><surname>Garc&#x00ED;a-Garc&#x00ED;a</surname><given-names>FJ</given-names> </name><name name-style="western"><surname>Rodr&#x00ED;guez-Ma&#x00F1;as</surname><given-names>L</given-names> </name><name name-style="western"><surname>Holzinger</surname><given-names>A</given-names> </name></person-group><article-title>Importance of medical data preprocessing in predictive modeling and risk factor discovery for the frailty syndrome</article-title><source>BMC Med Inform Decis Mak</source><year>2019</year><month>02</month><day>18</day><volume>19</volume><issue>1</issue><fpage>33</fpage><pub-id pub-id-type="doi">10.1186/s12911-019-0747-6</pub-id><pub-id pub-id-type="medline">30777059</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Maity</surname><given-names>S</given-names> </name><name name-style="western"><surname>Saikia</surname><given-names>MJ</given-names> </name></person-group><article-title>Large language models in healthcare and medical applications: a review</article-title><source>Bioengineering (Basel)</source><year>2025</year><month>06</month><day>10</day><volume>12</volume><issue>6</issue><fpage>631</fpage><pub-id pub-id-type="doi">10.3390/bioengineering12060631</pub-id><pub-id pub-id-type="medline">40564447</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Iqbal</surname><given-names>U</given-names> </name><name name-style="western"><surname>Tanweer</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rahmanti</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Greenfield</surname><given-names>D</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>LJ</given-names> </name><name name-style="western"><surname>Li</surname><given-names>YCJ</given-names> </name></person-group><article-title>Impact of large language model (ChatGPT) in healthcare: an umbrella review and evidence synthesis</article-title><source>J Biomed Sci</source><year>2025</year><month>05</month><day>7</day><volume>32</volume><issue>1</issue><fpage>45</fpage><pub-id pub-id-type="doi">10.1186/s12929-025-01131-z</pub-id><pub-id pub-id-type="medline">40335969</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhong</surname><given-names>W</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Performance of ChatGPT-4o and four open-source large language models in generating diagnoses based on China&#x2019;s rare disease catalog: comparative study</article-title><source>J Med Internet Res</source><year>2025</year><month>06</month><day>18</day><volume>27</volume><fpage>e69929</fpage><pub-id pub-id-type="doi">10.2196/69929</pub-id><pub-id pub-id-type="medline">40532199</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Park</surname><given-names>J</given-names> </name><name name-style="western"><surname>Patterson</surname><given-names>J</given-names> </name><name name-style="western"><surname>Acitores Cortina</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Gu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Hur</surname><given-names>C</given-names> </name><name name-style="western"><surname>Tatonetti</surname><given-names>N</given-names> </name></person-group><article-title>Enhancing EHR-based pancreatic cancer prediction with LLM-derived embeddings</article-title><source>NPJ Digit Med</source><year>2025</year><month>07</month><day>21</day><volume>8</volume><issue>1</issue><fpage>465</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01869-8</pub-id><pub-id pub-id-type="medline">40691317</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shool</surname><given-names>S</given-names> </name><name name-style="western"><surname>Adimi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Saboori Amleshi</surname><given-names>R</given-names> </name><name name-style="western"><surname>Bitaraf</surname><given-names>E</given-names> </name><name name-style="western"><surname>Golpira</surname><given-names>R</given-names> </name><name name-style="western"><surname>Tara</surname><given-names>M</given-names> </name></person-group><article-title>A systematic review of large language model (LLM) evaluations in clinical medicine</article-title><source>BMC Med Inform Decis Mak</source><year>2025</year><month>03</month><day>7</day><volume>25</volume><issue>1</issue><fpage>117</fpage><pub-id pub-id-type="doi">10.1186/s12911-025-02954-4</pub-id><pub-id pub-id-type="medline">40055694</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hennes</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Zeniya</surname><given-names>M</given-names> </name><name name-style="western"><surname>Czaja</surname><given-names>AJ</given-names> </name><etal/></person-group><article-title>Simplified criteria for the diagnosis of autoimmune hepatitis</article-title><source>Hepatology</source><year>2008</year><month>07</month><volume>48</volume><issue>1</issue><fpage>169</fpage><lpage>176</lpage><pub-id pub-id-type="doi">10.1002/hep.22322</pub-id><pub-id pub-id-type="medline">18537184</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hirschfield</surname><given-names>GM</given-names> </name><name name-style="western"><surname>Beuers</surname><given-names>U</given-names> </name><name name-style="western"><surname>Corpechot</surname><given-names>C</given-names> </name><etal/></person-group><article-title>EASL clinical practice guidelines: the diagnosis and management of patients with primary biliary cholangitis</article-title><source>J Hepatol</source><year>2017</year><month>07</month><volume>67</volume><issue>1</issue><fpage>145</fpage><lpage>172</lpage><pub-id pub-id-type="doi">10.1016/j.jhep.2017.03.022</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>You</surname><given-names>H</given-names> </name><name name-style="western"><surname>Duan</surname><given-names>W</given-names> </name><name name-style="western"><surname>Li</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Guidelines on the diagnosis and management of primary biliary cholangitis (2021)</article-title><source>J Clin Transl Hepatol</source><year>2023</year><month>06</month><day>28</day><volume>11</volume><issue>3</issue><fpage>736</fpage><lpage>746</lpage><pub-id pub-id-type="doi">10.14218/JCTH.2022.00347</pub-id><pub-id pub-id-type="medline">36969891</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Danan</surname><given-names>G</given-names> </name><name name-style="western"><surname>Teschke</surname><given-names>R</given-names> </name></person-group><article-title>RUCAM in drug and herb induced liver injury: the update</article-title><source>Int J Mol Sci</source><year>2015</year><month>12</month><day>24</day><volume>17</volume><issue>1</issue><fpage>14</fpage><pub-id pub-id-type="doi">10.3390/ijms17010014</pub-id><pub-id pub-id-type="medline">26712744</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>S</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Chinese guideline for the diagnosis and treatment of drug-induced liver injury: an update</article-title><source>Hepatol Int</source><year>2024</year><month>04</month><volume>18</volume><issue>2</issue><fpage>384</fpage><lpage>419</lpage><pub-id pub-id-type="doi">10.1007/s12072-023-10633-7</pub-id><pub-id pub-id-type="medline">38402364</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Terrault</surname><given-names>NA</given-names> </name><name name-style="western"><surname>Lok</surname><given-names>ASF</given-names> </name><name name-style="western"><surname>McMahon</surname><given-names>BJ</given-names> </name><etal/></person-group><article-title>Update on prevention, diagnosis, and treatment of chronic hepatitis B: AASLD 2018 hepatitis B guidance</article-title><source>Hepatology</source><year>2018</year><month>04</month><volume>67</volume><issue>4</issue><fpage>1560</fpage><lpage>1599</lpage><pub-id pub-id-type="doi">10.1002/hep.29800</pub-id><pub-id pub-id-type="medline">29405329</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>You</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>F</given-names> </name><name name-style="western"><surname>Li</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Guidelines for the prevention and treatment of chronic hepatitis B (version 2022)</article-title><source>J Clin Transl Hepatol</source><year>2023</year><month>11</month><day>28</day><volume>11</volume><issue>6</issue><fpage>1425</fpage><lpage>1442</lpage><pub-id pub-id-type="doi">10.14218/JCTH.2023.00320</pub-id><pub-id pub-id-type="medline">37719965</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hornung</surname><given-names>R</given-names> </name><name name-style="western"><surname>Bernau</surname><given-names>C</given-names> </name><name name-style="western"><surname>Truntzer</surname><given-names>C</given-names> </name><name name-style="western"><surname>Wilson</surname><given-names>R</given-names> </name><name name-style="western"><surname>Stadler</surname><given-names>T</given-names> </name><name name-style="western"><surname>Boulesteix</surname><given-names>AL</given-names> </name></person-group><article-title>A measure of the impact of CV incompleteness on prediction error estimation with application to PCA and normalization</article-title><source>BMC Med Res Methodol</source><year>2015</year><month>12</month><volume>15</volume><issue>1</issue><pub-id pub-id-type="doi">10.1186/s12874-015-0088-9</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Steyerberg</surname><given-names>EW</given-names> </name><name name-style="western"><surname>Harrell</surname><given-names>FE</given-names> </name></person-group><article-title>Prediction models need appropriate internal, internal-external, and external validation</article-title><source>J Clin Epidemiol</source><year>2016</year><month>01</month><volume>69</volume><fpage>245</fpage><lpage>247</lpage><pub-id pub-id-type="doi">10.1016/j.jclinepi.2015.04.005</pub-id><pub-id pub-id-type="medline">25981519</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cai</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Ji</surname><given-names>K</given-names> </name><etal/></person-group><article-title>HuatuoGPT-o1, towards medical complex reasoning with LLMs</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 25, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2412.18925</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="web"><article-title>Intelligent internet: II-medical-8B: medical reasoning model</article-title><source>Hugging Face</source><year>2025</year><access-date>2026-08-10</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://huggingface.co/Intelligent-Internet/II-Medical-8B">https://huggingface.co/Intelligent-Internet/II-Medical-8B</ext-link></comment></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>A</given-names> </name><name name-style="western"><surname>Li</surname><given-names>A</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Qwen3 technical report</article-title><source>arXiv</source><comment>Preprint posted online on  May 14, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2505.09388</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Sokolova</surname><given-names>M</given-names> </name></person-group><article-title>Specialists, scientists, and sentiments: word2Vec and doc2Vec in analysis of scientific and medical texts</article-title><source>SN Comput Sci</source><year>2021</year><volume>2</volume><issue>5</issue><fpage>414</fpage><pub-id pub-id-type="doi">10.1007/s42979-021-00807-1</pub-id><pub-id pub-id-type="medline">34414378</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ji</surname><given-names>H</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name></person-group><article-title>Early prediction of colorectal adenoma risk: leveraging large-language model for clinical electronic medical record data</article-title><source>Front Oncol</source><year>2025</year><month>05</month><day>15</day><volume>15</volume><fpage>1508455</fpage><pub-id pub-id-type="doi">10.3389/fonc.2025.1508455</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ramadan</surname><given-names>B</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>MC</given-names> </name><name name-style="western"><surname>Burkhart</surname><given-names>MC</given-names> </name><name name-style="western"><surname>Parker</surname><given-names>WF</given-names> </name><name name-style="western"><surname>Beaulieu-Jones</surname><given-names>BK</given-names> </name></person-group><article-title>Diagnostic codes in AI prediction models and label leakage of same-admission clinical outcomes</article-title><source>JAMA Netw Open</source><year>2025</year><month>12</month><day>1</day><volume>8</volume><issue>12</issue><fpage>e2550454</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2025.50454</pub-id><pub-id pub-id-type="medline">41632159</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vilakati</surname><given-names>S</given-names> </name></person-group><article-title>Prompt engineering for accurate statistical reasoning with large language models in medical research</article-title><source>Front Artif Intell</source><year>2025</year><volume>8</volume><fpage>1658316</fpage><pub-id pub-id-type="doi">10.3389/frai.2025.1658316</pub-id><pub-id pub-id-type="medline">41159127</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>B</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Langren&#x00E9;</surname><given-names>N</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>S</given-names> </name></person-group><article-title>Unleashing the potential of prompt engineering for large language models</article-title><source>Patterns (N Y)</source><year>2025</year><month>06</month><day>13</day><volume>6</volume><issue>6</issue><fpage>101260</fpage><pub-id pub-id-type="doi">10.1016/j.patter.2025.101260</pub-id><pub-id pub-id-type="medline">40575123</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sivarajkumar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kelley</surname><given-names>M</given-names> </name><name name-style="western"><surname>Samolyk-Mazzanti</surname><given-names>A</given-names> </name><name name-style="western"><surname>Visweswaran</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name></person-group><article-title>An empirical evaluation of prompting strategies for large language models in zero-shot clinical natural language processing: algorithm development and validation study</article-title><source>JMIR Med Inform</source><year>2024</year><month>04</month><day>8</day><volume>12</volume><fpage>e55318</fpage><pub-id pub-id-type="doi">10.2196/55318</pub-id><pub-id pub-id-type="medline">38587879</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gokseven Arda</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zeren Ozturk</surname><given-names>G</given-names> </name></person-group><article-title>Concordance of an artificial intelligence model (ChatGPT 4.0) with physician decisions in smoking cessation clinics: a comparative evaluation</article-title><source>Health Care (Don Mills)</source><year>2025</year><month>09</month><day>12</day><volume>13</volume><issue>18</issue><fpage>2283</fpage><pub-id pub-id-type="doi">10.3390/healthcare13182283</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>ClinicRealm: re-evaluating large language models with conventional machine learning for non-generative clinical prediction tasks</article-title><source>NPJ Digit Med</source><year>2026</year><month>04</month><day>8</day><volume>9</volume><issue>1</issue><fpage>319</fpage><pub-id pub-id-type="doi">10.1038/s41746-026-02539-z</pub-id><pub-id pub-id-type="medline">41951858</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kapoor</surname><given-names>S</given-names> </name><name name-style="western"><surname>Narayanan</surname><given-names>A</given-names> </name></person-group><article-title>Leakage and the reproducibility crisis in machine-learning-based science</article-title><source>Patterns (N Y)</source><year>2023</year><month>09</month><day>8</day><volume>4</volume><issue>9</issue><fpage>100804</fpage><pub-id pub-id-type="doi">10.1016/j.patter.2023.100804</pub-id><pub-id pub-id-type="medline">37720327</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schonlau</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zou</surname><given-names>RY</given-names> </name></person-group><article-title>The random forest algorithm for statistical learning</article-title><source>Stata J</source><year>2020</year><month>03</month><volume>20</volume><issue>1</issue><fpage>3</fpage><lpage>29</lpage><pub-id pub-id-type="doi">10.1177/1536867X20909688</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rumelhart</surname><given-names>DE</given-names> </name><name name-style="western"><surname>Hinton</surname><given-names>GE</given-names> </name><name name-style="western"><surname>Williams</surname><given-names>RJ</given-names> </name></person-group><article-title>Learning representations by back-propagating errors</article-title><source>Nature</source><year>1986</year><month>10</month><volume>323</volume><issue>6088</issue><fpage>533</fpage><lpage>536</lpage><pub-id pub-id-type="doi">10.1038/323533a0</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bewick</surname><given-names>V</given-names> </name><name name-style="western"><surname>Cheek</surname><given-names>L</given-names> </name><name name-style="western"><surname>Ball</surname><given-names>J</given-names> </name></person-group><article-title>Statistics review 14: logistic regression</article-title><source>Crit Care</source><year>2005</year><month>02</month><volume>9</volume><issue>1</issue><fpage>112</fpage><lpage>118</lpage><pub-id pub-id-type="doi">10.1186/cc3045</pub-id><pub-id pub-id-type="medline">15693993</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Guestrin</surname><given-names>C</given-names> </name></person-group><article-title>XGBoost: a scalable tree boosting system</article-title><year>2016</year><conf-name>The 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</conf-name><conf-date>Aug 13-17, 2016</conf-date><conf-loc>San Francisco, CA</conf-loc><fpage>785</fpage><lpage>794</lpage><pub-id pub-id-type="doi">10.1145/2939672.2939785</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cortes</surname><given-names>C</given-names> </name><name name-style="western"><surname>Vapnik</surname><given-names>V</given-names> </name></person-group><article-title>Support-vector networks</article-title><source>Mach Learn</source><year>1995</year><month>09</month><volume>20</volume><issue>3</issue><fpage>273</fpage><lpage>297</lpage><pub-id pub-id-type="doi">10.1007/BF00994018</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Das</surname><given-names>S</given-names> </name><name name-style="western"><surname>Nayak</surname><given-names>SP</given-names> </name><name name-style="western"><surname>Sahoo</surname><given-names>B</given-names> </name><name name-style="western"><surname>Champati Rai</surname><given-names>S</given-names> </name></person-group><article-title>A differential evolution-based optimized ensemble for balanced and imbalanced medical datasets</article-title><source>F1000Res</source><year>2025</year><volume>14</volume><fpage>1003</fpage><pub-id pub-id-type="doi">10.12688/f1000research.169456.2</pub-id><pub-id pub-id-type="medline">41694647</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tanaka</surname><given-names>A</given-names> </name><name name-style="western"><surname>Notohara</surname><given-names>K</given-names> </name><name name-style="western"><surname>Tobari</surname><given-names>M</given-names> </name><etal/></person-group><article-title>A clinicopathological study of IgG4-related autoimmune hepatitis and IgG4-hepatopathy</article-title><source>J Gastroenterol</source><year>2025</year><month>05</month><volume>60</volume><issue>5</issue><fpage>632</fpage><lpage>640</lpage><pub-id pub-id-type="doi">10.1007/s00535-025-02221-3</pub-id><pub-id pub-id-type="medline">39921744</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shah</surname><given-names>SK</given-names> </name><name name-style="western"><surname>Bowlus</surname><given-names>CL</given-names> </name></person-group><article-title>Autoimmune markers in primary biliary cholangitis</article-title><source>Clin Liver Dis</source><year>2024</year><month>02</month><volume>28</volume><issue>1</issue><fpage>93</fpage><lpage>101</lpage><pub-id pub-id-type="doi">10.1016/j.cld.2023.07.002</pub-id><pub-id pub-id-type="medline">37945165</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Prediction of primary biliary cholangitis among health check-up population with anti-mitochondrial M2 antibody positive</article-title><source>Clin Mol Hepatol</source><year>2025</year><month>04</month><volume>31</volume><issue>2</issue><fpage>474</fpage><lpage>488</lpage><pub-id pub-id-type="doi">10.3350/cmh.2024.0416</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bonino</surname><given-names>F</given-names> </name><name name-style="western"><surname>Colombatto</surname><given-names>P</given-names> </name><name name-style="western"><surname>Brunetto</surname><given-names>MR</given-names> </name></person-group><article-title>HBeAg-negative/anti-HBe-positive chronic hepatitis B: a 40-year-old history</article-title><source>Viruses</source><year>2022</year><month>07</month><day>30</day><volume>14</volume><issue>8</issue><fpage>1691</fpage><pub-id pub-id-type="doi">10.3390/v14081691</pub-id><pub-id pub-id-type="medline">36016312</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Soria</surname><given-names>A</given-names> </name><name name-style="western"><surname>D&#x00ED;az</surname><given-names>A</given-names> </name><name name-style="western"><surname>Iruzubieta</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Autoantibodies are associated with worse outcomes in MASLD</article-title><source>JHEP Rep</source><year>2025</year><month>10</month><volume>7</volume><issue>10</issue><fpage>101470</fpage><pub-id pub-id-type="doi">10.1016/j.jhepr.2025.101470</pub-id><pub-id pub-id-type="medline">40980159</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Terziroli Beretta-Piccoli</surname><given-names>B</given-names> </name><name name-style="western"><surname>Mieli-Vergani</surname><given-names>G</given-names> </name><name name-style="western"><surname>Vergani</surname><given-names>D</given-names> </name></person-group><article-title>Autoimmune hepatitis: serum autoantibodies in clinical practice</article-title><source>Clin Rev Allergy Immunol</source><year>2022</year><month>10</month><volume>63</volume><issue>2</issue><fpage>124</fpage><lpage>137</lpage><pub-id pub-id-type="doi">10.1007/s12016-021-08888-9</pub-id><pub-id pub-id-type="medline">34491531</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>K</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Pan</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Noninvasive diagnosis of AIH/PBC overlap syndrome based on prediction models</article-title><source>Open Med</source><year>2022</year><month>03</month><day>28</day><volume>17</volume><issue>1</issue><fpage>1550</fpage><lpage>1558</lpage><pub-id pub-id-type="doi">10.1515/med-2022-0526</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>L</given-names> </name><name name-style="western"><surname>Ji</surname><given-names>P</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>Y</given-names> </name></person-group><article-title>Machine learning model for hepatitis C diagnosis customized to each patient</article-title><source>IEEE Access</source><year>2022</year><volume>10</volume><fpage>106655</fpage><lpage>106672</lpage><pub-id pub-id-type="doi">10.1109/ACCESS.2022.3210347</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Balasubramanian</surname><given-names>JB</given-names> </name><name name-style="western"><surname>Adams</surname><given-names>D</given-names> </name><name name-style="western"><surname>Roxanis</surname><given-names>I</given-names> </name><etal/></person-group><article-title>Leveraging large language models for structured information extraction from pathology reports</article-title><source>J Pathol Inform</source><year>2025</year><month>11</month><volume>19</volume><fpage>100521</fpage><pub-id pub-id-type="doi">10.1016/j.jpi.2025.100521</pub-id><pub-id pub-id-type="medline">41340635</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Johnson</surname><given-names>B</given-names> </name><name name-style="western"><surname>Bath</surname><given-names>T</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Large language models for extracting histopathologic diagnoses of colorectal cancer and dysplasia from electronic health records</article-title><source>BMJ Open Gastroenterol</source><year>2025</year><month>09</month><day>18</day><volume>12</volume><issue>1</issue><fpage>e001896</fpage><pub-id pub-id-type="doi">10.1136/bmjgast-2025-001896</pub-id><pub-id pub-id-type="medline">40973184</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Li</surname><given-names>X</given-names> </name><name name-style="western"><surname>Fang</surname><given-names>K</given-names> </name><etal/></person-group><article-title>LLM-based-liverdisease-model: code repository for large language models for heterogeneous data mining in liver disease</article-title><source>GitHub</source><year>2025</year><access-date>2026-08-12</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/BioInfor-coder/LLM-based-LiverDisease-Model">https://github.com/BioInfor-coder/LLM-based-LiverDisease-Model</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>This supplementary material summarizes the basic information of three 8B-parameter large language models (Huatuo-o1, II-Medical, and Qwen3), including parameter size, release date, model type, maximum context length, underlying architecture, typical use cases, and the exact repository identifier and checkpoint or version tag used for local deployment.</p><media xlink:href="jmir_v28i1e92921_app1.xlsx" xlink:title="XLSX File, 9 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>This supplementary material provides the list of high-risk keywords that may result in label leakage.</p><media xlink:href="jmir_v28i1e92921_app2.xlsx" xlink:title="XLSX File, 14 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>This supplementary material provides an illustrative example of the standardized clinical text input used for large language model encoding.</p><media xlink:href="jmir_v28i1e92921_app3.docx" xlink:title="DOCX File, 16 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>This supplementary material shows a structured prompt template that assigns the large language model a hepatology assistant role and defines standardized clinical inputs and output format for differentiating liver disease.</p><media xlink:href="jmir_v28i1e92921_app4.png" xlink:title="PNG File, 264 KB"/></supplementary-material><supplementary-material id="app5"><label>Multimedia Appendix 5</label><p>This supplementary material contains the parameters for large language model inference.</p><media xlink:href="jmir_v28i1e92921_app5.xlsx" xlink:title="XLSX File, 9 KB"/></supplementary-material><supplementary-material id="app6"><label>Multimedia Appendix 6</label><p>This supplementary material includes 5-fold cross-validation grid search parameters.</p><media xlink:href="jmir_v28i1e92921_app6.xlsx" xlink:title="XLSX File, 9 KB"/></supplementary-material><supplementary-material id="app7"><label>Multimedia Appendix 7</label><p>This supplementary material contains the baseline demographics and comprehensive clinical laboratory characteristics of the study cohort.</p><media xlink:href="jmir_v28i1e92921_app7.xlsx" xlink:title="XLSX File, 21 KB"/></supplementary-material><supplementary-material id="app8"><label>Multimedia Appendix 8</label><p>This supplementary material shows the 3-class classification performance of the machine learning (ML)-only model and large language model (LLM)&#x2013;embedding models on the internal holdout validation set. The gray line denotes the ML-only model using the full set of clinical laboratory variables, while colored bars denote LLM-embedding models using embeddings from Huatuo-o1 (blue), II-Medical (red), and Qwen3 (green). The Qwen3-embedding model reached an accuracy of 0.919 (95% CI 0.907-0.932).</p><media xlink:href="jmir_v28i1e92921_app8.png" xlink:title="PNG File, 200 KB"/></supplementary-material><supplementary-material id="app9"><label>Multimedia Appendix 9</label><p>This supplementary material includes the comparison of large language model embeddings and 2 lightweight natural language models for 3-class text classification.</p><media xlink:href="jmir_v28i1e92921_app9.xlsx" xlink:title="XLSX File, 10 KB"/></supplementary-material><supplementary-material id="app10"><label>Multimedia Appendix 10</label><p>This supplementary material reports the per-case end-to-end inference time of the machine learning (ML)&#x2013;only and large language model (LLM)&#x2013;integrated ML models, decomposed by pipeline stage (feature loading, tokenization, LLM-embedding extraction, dimensionality reduction, feature concatenation, and downstream classification). Timing was measured on 1000 randomly selected patients from the modeling dataset; values are reported in milliseconds per case as median, mean, SD, minimum, and maximum.</p><media xlink:href="jmir_v28i1e92921_app10.xlsx" xlink:title="XLSX File, 10 KB"/></supplementary-material><supplementary-material id="app11"><label>Multimedia Appendix 11</label><p>This supplementary material shows the receiver operating characteristic curves for 4-class liver disease classification of the machine learning (ML)&#x2013;only and large language model (LLM)&#x2013;integrated ML models on the internal holdout validation set. The 3 LLM-integrated models (Huatuo-o1, II-Medical, and Qwen3) achieved macroaverage area under the receiver operating characteristic curves (AUROCs) of 0.969-0.973, compared with 0.954 for the ML-only model. Autoimmune hepatitis discrimination improved (AUROC 0.922-0.933 vs 0.895), and class-wise AUROC varied less across diseases among the LLM-integrated models.</p><media xlink:href="jmir_v28i1e92921_app11.png" xlink:title="PNG File, 182 KB"/></supplementary-material><supplementary-material id="app12"><label>Multimedia Appendix 12</label><p>This supplementary material shows the 3-class and 4-class classification performance of the machine learning (ML)&#x2013;only and large language model&#x2013;integrated ML models under a temporal split sensitivity analysis, using cases diagnosed between 2010 and 2019 for training and cases diagnosed between 2020 and 2025 for testing.</p><media xlink:href="jmir_v28i1e92921_app12.xlsx" xlink:title="XLSX File, 10 KB"/></supplementary-material><supplementary-material id="app13"><label>Multimedia Appendix 13</label><p>This supplementary material shows the medication recommendation performance metrics across large language models.</p><media xlink:href="jmir_v28i1e92921_app13.xlsx" xlink:title="XLSX File, 8 KB"/></supplementary-material></app-group></back></article>