<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="review-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e90209</article-id><article-id pub-id-type="doi">10.2196/90209</article-id><article-categories><subj-group subj-group-type="heading"><subject>Review</subject></subj-group></article-categories><title-group><article-title>Diagnostic Performance of Machine Learning for Systemic Lupus Erythematosus: Systematic Review and Meta-Analysis</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Wang</surname><given-names>Bingduo</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wang</surname><given-names>Zichao</given-names></name><degrees>MPH</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Liu</surname><given-names>Yang</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Ge</surname><given-names>Fangfang</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Anesthesiology and Intensive Care, The First Affiliated Hospital, Zhejiang University School of Medicine</institution><addr-line>Hangzhou</addr-line><country>China</country></aff><aff id="aff2"><institution>Department of Epidemiology and Statistics, Institute of Basic Medical Sciences Chinese Academy of Medical Sciences, School of Basic Medicine Peking Union Medical College</institution><addr-line>Beijing</addr-line><country>China</country></aff><aff id="aff3"><institution>Department of Proctology, The First Affiliated Hospital of Xinjiang Medical University</institution><addr-line>Xinjiang</addr-line><country>China</country></aff><aff id="aff4"><institution>Department of Hematology, The First Affiliated Hospital of Zhengzhou University</institution><addr-line>No. 1, Jianshe East Road, Erqi District</addr-line><addr-line>Zhengzhou</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Brini</surname><given-names>Stefano</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Abedi</surname><given-names>Iraj</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Le</surname><given-names>Nguyen Quoc Khanh</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Fangfang Ge, PhD, Department of Hematology, The First Affiliated Hospital of Zhengzhou University, No. 1, Jianshe East Road, Erqi District, Zhengzhou, China, 86 15067607819; <email>Gefangfang3@gmail.com</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>4</day><month>9</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e90209</elocation-id><history><date date-type="received"><day>23</day><month>12</month><year>2025</year></date><date date-type="rev-recd"><day>14</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>15</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9;Bingduo Wang, Zichao Wang, Yang Liu, Fangfang Ge. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 4.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e90209"/><abstract><sec><title>Background</title><p>Early and accurate diagnosis of systemic lupus erythematosus (SLE) and its organ involvement is essential. Previous reviews of machine learning (ML) in SLE combined heterogeneous tasks and validation strategies and may have overinterpreted model performance.</p></sec><sec><title>Objective</title><p>This study evaluated the diagnostic performance of ML and deep learning (DL) models for 3 clinically distinct SLE-related tasks: SLE classification or diagnosis, lupus nephritis (LN) diagnosis, and neuropsychiatric systemic lupus erythematosus (NPSLE) discrimination. We also assessed methodological quality and certainty of evidence.</p></sec><sec sec-type="methods"><title>Methods</title><p>PubMed, Embase, Cochrane Library, Web of Science, and IEEE Xplore were searched from January 2014 to April 2026. Eligible peer-reviewed diagnostic accuracy studies developed or validated ML or DL models for 1 of the 3 prespecified tasks, used an accepted reference standard, and provided data for a 2&#x00D7;2 contingency table. Bivariate random-effects meta-analyses with the Hartung-Knapp-Sidik-Jonkman adjustment were used to pool sensitivity and specificity. We reported 95% prediction intervals (PIs), assessed risk of bias using the Quality Assessment of Diagnostic Accuracy Studies for Artificial Intelligence tool (QUADAS-AI; Viknesh Sounderajah [Imperial College London]), and evaluated certainty of evidence using the Grading of Recommendations Assessment, Development, and Evaluation framework for diagnostic test accuracy.</p></sec><sec sec-type="results"><title>Results</title><p>Twenty-nine studies were included: 17 for SLE classification, 5 for LN diagnosis, and 7 for NPSLE discrimination. In the primary task-stratified analysis, pooled sensitivity was 0.91 (95% CI 0.86-0.94; 95% PI 0.56-0.99), and pooled specificity was 0.94 (95% CI 0.91-0.96; 95% PI 0.69-0.99), with low heterogeneity (<italic>I</italic>&#x00B2;=23.9% and 22.9%, respectively). DL models showed a sensitivity of 0.93 and specificity of 0.95, compared with 0.88 and 0.94 for traditional ML models. Certainty of evidence was high for most analyses but low for LN diagnosis because of inconsistency and imprecision. All studies were retrospective, and only 9 of 29 (31%) performed independent external validation. Overall risk of bias was high or unclear in 22 of 29 (75.9%) studies. No study reported model calibration, decision-curve analysis, or net clinical benefit.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>ML models showed promising diagnostic accuracy across 3 distinct SLE-related tasks, but wide PIs, limited external validation, and pervasive risk of bias restrict conclusions about real-world generalizability. Prospective multicenter studies with standardized tasks and reference standards, independent external validation, and formal assessment of calibration and clinical utility are required before clinical implementation.</p></sec><sec><title>Trial Registration</title><p>PROSPERO CRD42024545109; https://www.crd.york.ac.uk/prospero/view/CRD42024545109</p></sec></abstract><kwd-group><kwd>machine learning</kwd><kwd>systemic lupus erythematosus</kwd><kwd>diagnostic test accuracy</kwd><kwd>systematic review</kwd><kwd>meta-analysis</kwd><kwd>artificial intelligence</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Systemic lupus erythematosus (SLE) is an autoimmune disease that significantly affects multiple systems and organs [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. It is clinically manifested as dermatitis, lupus nephritis (LN), polyarthritis, pericarditis, and neuropsychiatric dysfunction [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref4">4</xref>]. These manifestations during disease progression complicate the accurate diagnosis [<xref ref-type="bibr" rid="ref5">5</xref>]. SLE is characterized by immune complexes and the hyperreactivity of B cells and T cells, leading to loss of immune tolerance for circulating antibodies in affected patients [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. Additionally, typical autoantibodies in the serum are elevated 3 to 9 years before clinical symptoms appear, particularly anti&#x2013;double-stranded DNA, anti-Smith, and antiphospholipid antibodies [<xref ref-type="bibr" rid="ref8">8</xref>-<xref ref-type="bibr" rid="ref10">10</xref>]. Hence, these biomarkers have been listed in the latest 2023 European League Against Rheumatism (EULAR) and the American College of Rheumatology (ACR) criteria for SLE classification to guide thorough assessment to mitigate the risk of misdiagnosis [<xref ref-type="bibr" rid="ref11">11</xref>].</p><p>Despite the existing knowledge, a recent study conducted by Adamichou et al [<xref ref-type="bibr" rid="ref8">8</xref>] suggested that up to 20% of early cohort patients could not be diagnosed using the 2019 EULAR or ACR criteria, and the Systemic Lupus Collaborating Clinics (SLICC)-2012 criteria. However, the integration of EULAR or ACR and the SLICC criteria through machine learning (ML) considerably enhanced sensitivity (from 80% to 97%) for early cases [<xref ref-type="bibr" rid="ref12">12</xref>]. Notably, neuropsychiatric symptoms of lupus are frequently observed among pediatric patients [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref14">14</xref>]. However, clinical examinations of cerebrospinal fluid and serum and electroencephalograms (EEGs) all fail to diagnose early-onset SLE [<xref ref-type="bibr" rid="ref15">15</xref>]. Currently, it remains unclear whether a comprehensive analysis of all features can aid in the early diagnosis and timely treatment of these patients [<xref ref-type="bibr" rid="ref16">16</xref>]. ML can identify patterns in large, complex datasets and support diagnostic decision-making [<xref ref-type="bibr" rid="ref17">17</xref>].</p><p>Over the past 5 decades, since the initial discussions on AI in the medical field, ML and deep learning (DL) have made significant strides, enabling them to effectively learn from accumulated knowledge and statistical data [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>]. In addition to the advantages of automation and efficiency, AI-based algorithms enhance scalability and improve decision-making processes for health care systems [<xref ref-type="bibr" rid="ref21">21</xref>]. Recently, electronic medical records have been widely adopted, and biogenetic factors have become more intricate, which has made the need for advanced technological solutions more urgent than ever [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref22">22</xref>]. AI is emerging as the most suitable option to address this challenge [<xref ref-type="bibr" rid="ref23">23</xref>]. Interestingly, Wu et al [<xref ref-type="bibr" rid="ref14">14</xref>] developed a diagnostic model to classify metastasis stages of bladder cancer, with nearly 100% sensitivity. Therefore, given the high expectation for AI, its application in disease detection and classification is imperative, as it has the potential to revolutionize conventional clinical diagnostic approaches [<xref ref-type="bibr" rid="ref24">24</xref>].</p><p>To ground this work in the broader context of AI in medical diagnostics, we build on recent advances in the field: AI-driven medical image analysis has demonstrated robust diagnostic performance across a wide range of clinical indications, with DL models proving particularly effective at extracting complex patterns from multimodal medical data [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>]. DL has also been validated for standardized assessment of autoimmune disease-related imaging, reducing interreader variability and improving diagnostic efficiency in rheumatology [<xref ref-type="bibr" rid="ref27">27</xref>]. These advances provide a strong foundational rationale for exploring ML for SLE diagnosis, but also highlight the need for rigorous, clinically oriented, and comprehensive analysis of the evidence to guide clinical translation [<xref ref-type="bibr" rid="ref28">28</xref>-<xref ref-type="bibr" rid="ref30">30</xref>].</p><p>Despite the rapid growth of research in this field, existing systematic reviews on the application of ML for SLE diagnosis have critical, fundamental limitations that undermine the validity of their conclusions [<xref ref-type="bibr" rid="ref31">31</xref>]. First, and most importantly, prior syntheses have pooled heterogeneous, clinically incomparable tasks&#x2014;including SLE classification, LN activity grading, neuropsychiatric systemic lupus erythematosus (NPSLE) vs multiple sclerosis discrimination, and disease activity score estimation&#x2014;into a single meta-analysis, which directly violates the core assumptions of diagnostic test accuracy (DTA) meta-analysis and results in pooled estimates that lack clinical interpretability [<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>]. Second, previous reviews have conflated results from internal validation (which is well known to systematically overestimate model performance) and independent external validation, leading to an overestimation of real-world generalizability [<xref ref-type="bibr" rid="ref34">34</xref>]. Third, no prior review has used state-of-the-art statistical methods for DTA meta-analysis, including the Hartung-Knapp-Sidik-Jonkman (HKSJ) random-effects model and prediction intervals (PIs); these methods are essential for addressing extreme between-study heterogeneity and quantifying real-world performance variability [<xref ref-type="bibr" rid="ref35">35</xref>]. Fourth, existing syntheses have not yet used the QUADAS-AI (Quality Assessment of Diagnostic Accuracy Studies for Artificial Intelligence) tool (Viknesh Sounderajah [Imperial College London]), the validated gold standard for AI-based diagnostic studies, to comprehensively assess the methodological quality of included studies, nor have they addressed the widespread risk of optimism bias from selective reporting of best-performing models [<xref ref-type="bibr" rid="ref36">36</xref>]. Finally, no review has systematically evaluated the critical gap in clinical utility assessment, including model calibration, net clinical benefit, and workflow integration, which are essential to determine whether ML models will improve patient outcomes in real-world clinical settings [<xref ref-type="bibr" rid="ref37">37</xref>].</p><p>This systematic review and meta-analysis addresses all of these critical gaps in existing literature, with 4 core innovations and contributions to the field. First, with respect to methodological rigor, we perform the first fully task-stratified DTA meta-analysis, with primary analyses restricted to 3 independent, clinically homogeneous diagnostic tasks (SLE classification, LN diagnosis, and NPSLE discrimination), eliminating the core flaw of heterogeneous task pooling that invalidated prior reviews. Second, with respect to statistical methodology, we use the HKSJ random-effects model (recommended for DTA meta-analysis) and 95% PIs to quantify real-world performance variability, reporting PI lines in all forest plots and avoiding overinterpretation of pooled point estimates in the context of heterogeneity. Third, with respect to bias mitigation, we implement a strict, prespecified rule for contingency table inclusion, with each study contributing only one independent table to primary analyses (prioritizing external validation and prespecified model results, rather than post hoc best-performing models), eliminating data dependency, double counting, and optimism bias. Fourth, with respect to clinical relevance, we systematically distinguish internal versus external validation results, comprehensively appraise study quality with QUADAS-AI, perform GRADE (Grading of Recommendations Assessment, Development, and Evaluation) assessment of certainty of evidence, and explicitly address the absence of clinical utility assessment in existing research, providing evidence-based guidance for both future research and clinical implementation.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Protocol and Registration</title><p>This systematic review and meta-analysis was conducted in strict accordance with the PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) 2020 statement (<xref ref-type="supplementary-material" rid="app5">Checklist 1</xref>), PRISMA-DTA (PRISMA for Diagnostic Test Accuracy) guidelines, and PRISMA-S (Preferred Reporting Items for Systematic Reviews and Meta-Analyses literature search extension) reporting standards. The study protocol was preregistered in the International Prospective Register of Systematic Reviews (PROSPERO) on May 7, 2024 (registration number: CRD42024545109). All deviations from the preregistered protocol are explicitly listed in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>, with a rationale for each change. Ethical approval and informed consent were not required for this systematic review of published literature [<xref ref-type="bibr" rid="ref15">15</xref>]. The completed PRISMA 2020, PRISMA-DTA, and PRISMA-S checklists are provided in <xref ref-type="supplementary-material" rid="app5">Checklist 1</xref><xref ref-type="supplementary-material" rid="app6"/><xref ref-type="supplementary-material" rid="app7"/>-<xref ref-type="supplementary-material" rid="app8">4</xref>. The following PRISMA-S items were not applicable to this review: (1) item 12 (automated search alerts or updates): no automated alerts were set, as the search was rerun manually at the time of revision; (2) item 13 (gray literature and other resources): the review was restricted to full-text peer-reviewed publications, and no gray literature, trial registries, or unpublished sources were searched; and (3) item 14 (contacting authors for unpublished data): authors were not contacted, as only published diagnostic accuracy data permitting construction of 2&#x00D7;2 contingency tables were eligible for inclusion. These decisions are consistent with the preregistered study protocol (PROSPERO CRD42024545109).</p></sec><sec id="s2-2"><title>Literature Search Strategy</title><p>A comprehensive, peer-reviewed literature search was developed in collaboration with a medical librarian, in full adherence to the Cochrane Handbook for Systematic Reviews of Diagnostic Test Accuracy and PRISMA-S guidelines. The search was originally conducted through January 2025 and was fully updated through April 2026 at the time of revision. PubMed, Embase, Cochrane Library, Web of Science, and IEEE Xplore were searched for studies published between January 1, 2014, and April 2026. No geographical restrictions were applied, and only studies published in English were included.</p><p>The search strategy combined MeSH terms, Emtree terms, and free-text keywords for three core concepts: (1) SLE and related manifestations, (2) ML and AI, and (3) DTA. Field modifiers were used to restrict terms to the title or abstract fields, where appropriate, to maximize sensitivity and specificity. The full, unedited search strategy for each database is provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p><p>Additional studies were identified through manual screening of the reference lists of included studies and relevant systematic reviews, to ensure that no eligible studies were missed.</p></sec><sec id="s2-3"><title>Study Selection Criteria</title><p>Studies were eligible if they met all of the following criteria. First, regarding study design, eligible studies were full-text, peer-reviewed diagnostic accuracy studies that developed or validated ML or DL models for 1 of the 3 prespecified, clinically homogeneous diagnostic tasks (SLE classification, LN diagnosis, or NPSLE discrimination). Second, regarding the index test, eligible studies used ML or DL models with any type of clinical input data, including histopathology images, radiological images (magnetic resonance imaging [MRI], computed tomography [CT], and ultrasound), spectroscopic data (Raman spectroscopy and Fourier transform infrared spectroscopy [FTIR]), and electronic health record (EHR) data. Third, regarding the reference standard, a clinically accepted independent reference standard was required: for SLE classification, the 1997 ACR, 2012 SLICC, or 2019 EULAR or ACR criteria, or expert consensus by board-certified rheumatologists; for LN diagnosis, renal biopsy (International Society of Nephrology [ISN] or Renal Pathology Society [RPS] classification); and for NPSLE discrimination, the 1999 ACR NPSLE definitions with expert consensus. Fourth, regarding outcome data, studies were required to report sufficient information to construct a 2&#x00D7;2 contingency table (true positive [TP], false positive [FP], false negative [FN], and true negative [TN]) or to provide sensitivity, specificity, and sample size to allow calculation of these values.</p><p>Studies were excluded if they met any of the following criteria: nonhuman or animal studies; review articles, editorials, case reports, conference abstracts, letters, or study protocols; studies focused solely on disease prognosis, treatment response prediction, or disease activity score estimation without a primary diagnostic outcome; studies that did not use an independent reference standard or where the reference standard was partially derived from the ML model input features (circular validation); duplicate publications of the same study cohort; or studies with a sample size of fewer than 20 participants.</p><p>Two independent reviewers (BW and ZW) performed title and abstract screening in duplicate, followed by full-text screening of potentially eligible studies. Disagreements were resolved by consensus with a third senior reviewer (FG).</p></sec><sec id="s2-4"><title>Data Extraction</title><p>Data extraction was performed independently by 3 reviewers (BW, ZW, and YL) using a prepiloted, standardized data extraction form. The form was expanded to include items specific to AI-based diagnostic studies, and all extracted data were cross-checked for accuracy. Discrepancies were resolved by consensus with a fourth reviewer (FG).</p><p>The following data were extracted from each included study. Study characteristics included first author, year of publication, study country, study design, study period, clinical setting, sample size, participant demographics, and prespecified inclusion and exclusion criteria. Clinical task details included the explicit definition of the diagnostic task, target population, and clinical decision context. Index test details included the type of ML or DL algorithm, input data modality, model preprocessing steps, training, validation, and testing workflow, and whether the reported model was prespecified or post hoc selected as the best-performing model. Validation strategy details included clear definitions of training, validation, and test sets; whether patient-level splits were used; whether internal or independent external validation was performed; and whether data leakage was assessed and excluded. Reference standard details included the type of reference standard used, definition of positive and negative cases, and whether the reference standard assessment was blinded to the index test results. Diagnostic accuracy outcomes included TP, FP, FN, TN, sensitivity, specificity, area under the curve (AUC), positive predictive value (PPV), negative predictive value (NPV), and 95% CIs, with separate extraction for internal and external validation cohorts. Secondary outcomes included head-to-head comparisons with human clinicians, model calibration, decision-curve analysis (DCA), net clinical benefit, and implementation outcomes. Risk of bias items included all items required for the QUADAS-AI risk of bias assessment.</p></sec><sec id="s2-5"><title>Rules for Including Contingency Tables (Prespecified to Eliminate Bias and Data Dependency)</title><p>To avoid double-counting, data dependency, and optimism bias, we implemented the following strict, prespecified rules for including contingency tables in meta-analyses. The following prespecified rules governed contingency table inclusion. For the primary meta-analyses, each independent study cohort contributed exactly one 2&#x00D7;2 contingency table, selected according to the following priority order: first, results from an independent external validation cohort for a prespecified model; second, results from an internal validation cohort for a prespecified model; third, results from the full study cohort for a prespecified model. Post hoc selected best-performing model results were explicitly excluded from all primary analyses to eliminate researcher-driven optimism bias. For subgroup and sensitivity analyses, multiple nonoverlapping contingency tables from the same study were included only if they derived from mutually exclusive patient cohorts with no overlapping participants; all nonindependent data were explicitly labeled and their limitations acknowledged in the paper. Two levels of exploratory analyses were conducted: (1) an exploratory all-task pooled analysis using one table per study (selected using the priority order above) to provide an overall summary, with a clear statement that this analysis has no clinical interpretability; and (2) an exploratory all-table analysis in which all eligible algorithm-level contingency tables from each study were included (124 tables from 29 studies [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref45">45</xref>]), explicitly treated as exploratory because multiple tables could originate from the same study, potentially introducing within-study correlation.</p></sec><sec id="s2-6"><title>Risk of Bias Assessment</title><p>The risk of bias and applicability concerns of each included study were assessed independently by 2 reviewers (BW and FG) using the QUADAS-AI tool, the validated gold standard for AI-based diagnostic accuracy studies. The QUADAS-AI tool comprises 4 domains: patient selection, index test, reference standard, and flow and timing. Each domain was rated for risk of bias (low, high, or unclear) and applicability concerns (low, high, or unclear), with explicit justifications for each rating. Disagreements were resolved by consensus with a third reviewer (YL).</p></sec><sec id="s2-7"><title>Statistical Analysis</title><p>All statistical analyses were prespecified in the study protocol and performed using R (version 4.4.2; R Core Team) with the mada package for bivariate random-effects meta-analyses. Statistical significance was assessed using a 2-sided &#x03B1; level of .05.</p></sec><sec id="s2-8"><title>Primary Meta-Analyses</title><p>For each of the 3 independent clinical tasks, we performed hierarchical summary receiver operating characteristic (HSROC) meta-analyses to estimate pooled sensitivity, specificity, and AUC. The HKSJ random-effects model was used for all pooled analyses, as this method provides more robust type I error control and more precise effect estimates than the standard DerSimonian-Laird method, particularly in the setting of high between-study heterogeneity and small numbers of included studies.</p><p>For all pooled sensitivity and specificity estimates, we reported the pooled point estimate, 95% CI, which quantifies uncertainty around the average effect, and 95% PI, which quantifies the expected range of true effect sizes across different settings for 95% of future similar studies, and provides a clinically relevant estimate of real-world performance variability. The CI and PI serve fundamentally different purposes. The CI reflects the precision of the pooled mean, whereas the PI reflects the distribution of expected effects in new settings, accounting for between-study variance [<xref ref-type="bibr" rid="ref35">35</xref>].</p><p>Between-study heterogeneity was quantified using the <italic>I</italic>&#x00B2; statistic, with the following thresholds: <italic>I</italic>&#x00B2;&#x003C;50%= low heterogeneity, 50%&#x2010;75%=moderate heterogeneity,&#x003E;75%= high heterogeneity,&#x003E;90%= extreme heterogeneity. The <italic>I</italic>&#x00B2; statistic has limited utility in practical applications because it cannot quantify the magnitude of true effect variation across populations. Therefore, the 95% PI is used as the primary measure of real-world heterogeneity. For analyses with extreme heterogeneity (<italic>I</italic>&#x00B2;&#x003E;90%), the pooled point estimate has limited clinical significance, and the PI should be used to interpret real-world performance.</p></sec><sec id="s2-9"><title>GRADE Certainty-of-Evidence Assessment</title><p>The certainty of evidence for each analysis was evaluated using the GRADE framework adapted for DTA studies [<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref47">47</xref>]. The certainty started at high and was rated down for five domains: (1) risk of bias, downgraded if &#x2265;1/3 of studies had any QUADAS-AI domain rated as high risk; (2) inconsistency, downgraded one level if <italic>I</italic>&#x00B2;&#x003E;50% and 2 levels if <italic>I</italic>&#x00B2;&#x003E;75% for either sensitivity or specificity; (3) indirectness, downgraded if the study populations, index tests, or reference standards did not directly match the review question; (4) imprecision, downgraded if the number of studies was&#x003C;5 or the maximum 95% CI width for pooled sensitivity or specificity exceeded 0.20; and (5) publication bias, downgraded if Deeks asymmetry test met the prespecified significance threshold of &#x03B1;=.10. The GRADE assessment was performed for the overall analysis, ML and DL subgroups, each clinical task, and the external validation subgroup.</p></sec><sec id="s2-10"><title>Subgroup and Sensitivity Analyses</title><p>Prespecified subgroup analyses were performed to explore potential sources of heterogeneity, using the HKSJ model, across the following dimensions: validation strategy (internal validation vs external validation); algorithm type (traditional ML vs DL); input data modality (histopathology vs radiological imaging vs spectroscopic data vs EHR data); sample size (&#x2265;100 participants vs &#x003C;100 participants); and risk of bias (low overall risk vs high or unclear overall risk).</p><p>Prespecified sensitivity analyses were performed using the leave-one-out analysis (sequentially removing each study to assess its impact on the pooled estimate), by excluding studies with a high overall risk of bias, and by excluding studies with a sample size&#x003C;50 participants to assess the robustness of the primary pooled estimates.</p></sec><sec id="s2-11"><title>Small-Study Effects Assessment</title><p>Visual inspection of funnel plots and the Deeks funnel plot asymmetry test were used to assess small-study effects. These methods can only assess small-study effects, not publication bias, as publication bias is only one of the many potential causes of small-study effects. Additionally, these analyses have limited statistical power with a small number of included studies and cannot rule out researcher-driven optimism bias from selective reporting of favorable results.</p></sec><sec id="s2-12"><title>Comparison Between ML Models and Human Clinicians</title><p>Given that only 2 studies [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref40">40</xref>] reported head-to-head comparisons between ML models and human clinicians, with sparse data, heterogeneous clinical tasks, and varying clinician expertise levels, we performed an exploratory descriptive summary receiver operating characteristic (SROC) analysis only to visualize the comparative diagnostic space; no inferential meta-analysis or formal between-group statistical comparison was attempted, as the data were insufficient to produce reliable pooled estimates.</p></sec><sec id="s2-13"><title>Ethical Considerations</title><p>As this is a systematic review and meta-analysis of published literature, ethical approval and informed consent are not applicable.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Study Selection</title><p>A total of 6872 records were retrieved from the electronic database search, with 1159 duplicate records removed, leaving 5713 records for screening. Of these, 5411 records were excluded during initial title and abstract screening, leaving 302 potentially relevant records. A further 195 records were excluded during secondary eligibility screening before full-text retrieval, leaving 107 reports sought for retrieval. All 107 reports were retrieved. Sixty-three reports were excluded during preliminary full-text triage before formal eligibility assessment, leaving 44 reports assessed in detail. Of these, 15 were excluded for the following reasons: not a diagnostic accuracy study (n=12), incomplete data for construction of a 2&#x00D7;2 table (n=2), and research topic mismatch (n=1). The remaining 29 eligible studies comprised 25 studies included in the previous version of the review [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref41">41</xref>] and 4 newly included studies [<xref ref-type="bibr" rid="ref42">42</xref>-<xref ref-type="bibr" rid="ref45">45</xref>], yielding a total of 29 studies in the updated review. The detailed study selection process is shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><p>A total of 29 studies met all criteria for the quantitative meta-analysis [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref45">45</xref>]. Aliyev et al [<xref ref-type="bibr" rid="ref48">48</xref>] and de Ara&#x00FA;jo et al [<xref ref-type="bibr" rid="ref33">33</xref>] were excluded from all analyses. The 29 studies were distributed across the 3 prespecified clinical tasks as follows: 17 studies for SLE classification [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref29">29</xref>-<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref40">40</xref>-<xref ref-type="bibr" rid="ref44">44</xref>], 5 studies for LN diagnosis [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref32">32</xref>], and 7 studies for NPSLE discrimination [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref45">45</xref>], with no overlapping patient cohorts across tasks. In the all-table exploratory analysis, 124 tables from 29 studies were included [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref45">45</xref>]. The exploratory receiver operating characteristic (ROC) results stratified by task type and disease type are shown in <xref ref-type="fig" rid="figure2">Figures 2</xref> and <xref ref-type="fig" rid="figure3">3</xref>, respectively. The pooled overall ROC analysis across all included studies is presented in Figure S1 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>. The forest plot for the all-table exploratory analysis is presented in Figure S3 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) 2020 flow diagram of study selection.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e90209_fig01.png"/></fig><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Pooled performance stratified by task type (exploratory analysis). (A) Receiver operating characteristic curves for systemic lupus erythematosus detection (22 studies with 90 tables) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref29">29</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref40">40</xref>-<xref ref-type="bibr" rid="ref44">44</xref>] and (B) receiver operating characteristic curves for systemic lupus erythematosus classification (7 studies with 34 tables) [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref45">45</xref>]. AUC: area under the curve; SENS: sensitivity; SLE: systemic lupus erythematosus; SPEC: specificity; SROC: summary receiver operating characteristic.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e90209_fig02.png"/></fig><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Pooled performance stratified by disease type (exploratory analysis). (A) Systemic lupus erythematosus (17 studies with 65 tables) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref29">29</xref>-<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref40">40</xref>-<xref ref-type="bibr" rid="ref44">44</xref>], (B) lupus nephritis (5 studies with 25 tables) [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref32">32</xref>], and (C) neuropsychiatric systemic lupus erythematosus (7 studies with 34 tables) [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref45">45</xref>]. AUC: area under the curve; LN: lupus nephritis; NPSLE: neuropsychiatric systemic lupus erythematosus; SENS: sensitivity; SLE: systemic lupus erythematosus; SPEC: specificity; SROC: summary receiver operating characteristic.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e90209_fig03.png"/></fig></sec><sec id="s3-2"><title>Study Characteristics</title><p>The baseline characteristics of the 29 included studies [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref45">45</xref>] are summarized in Tables S1-S3 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>. The studies were published between 2018 and 2026, with sample sizes ranging from 26 to 6476 participants. All included studies were retrospective in design. Of the 29 studies [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref45">45</xref>], 9 (31%) [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref45">45</xref>] performed independent external validation using out-of-sample, multicenter, or temporally separated cohorts; the remaining 20 (69%) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref25">25</xref>-<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref29">29</xref>-<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref44">44</xref>] used only internal validation (random split-sample or k-fold cross-validation within a single dataset). Study design and basic demographic characteristics are summarized in Table S3 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>; the methods of model training and validation are summarized in Table S1 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>; indicators, algorithms, and data sources are summarized in Table S2 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>; terminology mapping across included studies is provided in Table S4 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>; and the implementation and reporting checklist is provided in Table S5 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p><p>The included studies covered 3 primary clinical tasks, including SLE classification or diagnosis (17 studies) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref29">29</xref>-<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref40">40</xref>-<xref ref-type="bibr" rid="ref45">45</xref>], LN diagnosis and classification (5 studies) [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref32">32</xref>], and NPSLE discrimination (7 studies) [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref45">45</xref>]. Input data modalities included histopathology images (9 studies, comprising 7 independent patient cohorts) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref20">20</xref>-<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref37">37</xref>], MRI (8 studies) [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref45">45</xref>], Raman spectroscopy (3 studies) [<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref41">41</xref>], clinical images (3 studies) [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref40">40</xref>], retinal fundus imaging (1 study) [<xref ref-type="bibr" rid="ref42">42</xref>], nailfold videocapillaroscopy (1 study) [<xref ref-type="bibr" rid="ref44">44</xref>], optical coherence tomography (OCT; 1 study) [<xref ref-type="bibr" rid="ref31">31</xref>], ultrasound (1 study) [<xref ref-type="bibr" rid="ref26">26</xref>], finger optical diffusion imaging (FODI; 1 study) [<xref ref-type="bibr" rid="ref19">19</xref>], and EHR and laboratory data (1 study) [<xref ref-type="bibr" rid="ref43">43</xref>]. For subgroup meta-analysis by modality, retinal fundus imaging and nailfold videocapillaroscopy were grouped with clinical images (5 studies total) [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref44">44</xref>], and only modality subgroups represented by at least 3 studies were analyzed separately. Only 2 studies (6.9%) [<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref42">42</xref>] performed head-to-head comparisons between ML models and board-certified rheumatologists or dermatologists using the same held-out test dataset. None of the included studies reported model calibration, DCA, or net clinical benefit.</p></sec><sec id="s3-3"><title>Risk of Bias Assessment</title><p>The results of the QUADAS-AI risk of bias assessment are presented in <xref ref-type="fig" rid="figure4">Figures 4</xref> and <xref ref-type="fig" rid="figure5">5</xref>. Overall, 22 out of 29 studies (75.9%) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref24">24</xref>-<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref41">41</xref>-<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref45">45</xref>] were judged to have high or unclear overall risk of bias.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>QUADAS-AI (Quality Assessment of Diagnostic Accuracy Studies for Artificial Intelligence) summary risk of bias and applicability concerns.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e90209_fig04.png"/></fig><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>QUADAS-AI (Quality Assessment of Diagnostic Accuracy Studies for Artificial Intelligence) per-study traffic-light plot for risk of bias and applicability concerns [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref45">45</xref>].</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e90209_fig05.png"/></fig><p>In the domain of patient selection, 18 (62.1%) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref25">25</xref>-<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref41">41</xref>-<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref45">45</xref>] studies had high or unclear risk of bias, primarily due to nonconsecutive patient recruitment, case-control design that did not reflect the clinical target population, or inappropriate exclusion criteria. For the index test, 27 (93.1%) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref20">20</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref37">37</xref>-<xref ref-type="bibr" rid="ref45">45</xref>] studies had low risk of bias, while 1 (3.4%) [<xref ref-type="bibr" rid="ref36">36</xref>] study had unclear risk of bias and 1 (3.4%) study [<xref ref-type="bibr" rid="ref19">19</xref>] had high risk of bias. For reference standards, 24 (82.8%) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref20">20</xref>-<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref27">27</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref39">39</xref>-<xref ref-type="bibr" rid="ref45">45</xref>] studies had low risk of bias, 2 (6.9%) [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref38">38</xref>] studies had unclear risk of bias, and 3 (10.3%) [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>] studies had high risk of bias, primarily due to lack of blinding of the reference standard assessment to the index test results or inconsistent application of the reference standard. Regarding flow and timing, 19 (65.5%) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref41">41</xref>-<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref45">45</xref>] studies had high or unclear risk of bias, primarily due to lack of reporting of the time interval between the index test and reference standard, or exclusion of participants from the final analysis without justification.</p><p>For applicability concerns, 4 (13.8%) [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref44">44</xref>] studies had unclear concerns related to patient selection. For the index test, 2 (6.9%) [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref36">36</xref>] studies had unclear concerns. For the reference standard, 3 (10.3%) [<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref42">42</xref>] studies had unclear concerns, and 4 (13.8%) [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>] studies had high concerns.</p></sec><sec id="s3-4"><title>Meta-Analyses of Task-Stratified Diagnostic Performance</title><p>The HKSJ random-effects model was used for all primary analyses, with one independent contingency table per study, prioritizing external validation and prespecified model results. In the primary analysis of all 29 studies [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref45">45</xref>], the pooled sensitivity was 0.91 (95% CI 0.86&#x2010;0.94, 95% PI 0.56&#x2010;0.99) and the pooled specificity was 0.94 (95% CI 0.91&#x2010;0.96, 95% PI 0.69&#x2010;0.99), with low between-study heterogeneity (<italic>I</italic>&#x00B2;=23.9% for sensitivity, <italic>I</italic>&#x00B2;=22.9% for specificity). Deeks funnel plot asymmetry test showed no significant small-study effects (<italic>P</italic>=.19). The Cochrane-style forest plot for the primary analysis is shown in <xref ref-type="fig" rid="figure6">Figure 6</xref>. The forest plot for the primary task-stratified analysis is also presented in Figure S2 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Cochrane-style forest plot for the primary diagnostic meta-analysis (29 studies) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref45">45</xref>]. FN: false negative; FP: false positive; PI: prediction interval; SENS: sensitivity; SPEC: specificity; TN: true negative; TP: true positive; Wt: study weight. Random-effects bivariate model with Knapp-Hartung adjustment; dashed red=95% prediction interval.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e90209_fig06.png"/></fig><p>A total of 17 studies (one independent contingency table per study) were included in the primary analysis for SLE classification (Task 1) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref29">29</xref>-<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref40">40</xref>-<xref ref-type="bibr" rid="ref44">44</xref>]. The pooled results showed a sensitivity of 0.90 (95% CI 0.84&#x2010;0.94, 95% PI 0.53&#x2010;0.99) and specificity of 0.94 (95% CI 0.92&#x2010;0.96, 95% PI 0.85&#x2010;0.98). Low between-study heterogeneity (<italic>I</italic>&#x00B2;=24.7% for sensitivity, <italic>I</italic>&#x00B2;=6.7% for specificity) was observed. The CI, reflecting precision of the average effect, was relatively narrow, whereas the PI, reflecting the expected distribution of performance in new settings, was considerably wider. This distinction is critical: the narrow CI indicates that the average performance is precisely estimated, while the wide PI (sensitivity 0.53&#x2010;0.99) indicates that a new study conducted in a different setting could yield substantially different results depending on population characteristics, imaging modality, and reference standard [<xref ref-type="bibr" rid="ref35">35</xref>]. According to the GRADE assessment, the certainty of evidence was high for SLE classification (<xref ref-type="table" rid="table1">Table 1</xref>).</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>GRADE<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> certainty of evidence for diagnostic test accuracy (29 studies) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref45">45</xref>].</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Outcome</td><td align="left" valign="bottom">Number of studies</td><td align="left" valign="bottom">Risk of bias</td><td align="left" valign="bottom">Inconsistency</td><td align="left" valign="bottom">Indirectness</td><td align="left" valign="bottom">Imprecision</td><td align="left" valign="bottom">Publication bias</td><td align="left" valign="bottom">Certainty</td></tr></thead><tbody><tr><td align="left" valign="top">Overall (29 studies)</td><td align="left" valign="top">29</td><td align="left" valign="top">No serious</td><td align="left" valign="top">No serious</td><td align="left" valign="top">No serious</td><td align="left" valign="top">No serious</td><td align="left" valign="top">Undetected</td><td align="left" valign="top">High</td></tr><tr><td align="left" valign="top">ML<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> subgroup</td><td align="left" valign="top">19</td><td align="left" valign="top">No serious</td><td align="left" valign="top">No serious</td><td align="left" valign="top">No serious</td><td align="left" valign="top">No serious</td><td align="left" valign="top">Not assessed</td><td align="left" valign="top">High</td></tr><tr><td align="left" valign="top">DL<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup> subgroup</td><td align="left" valign="top">10</td><td align="left" valign="top">No serious</td><td align="left" valign="top">No serious</td><td align="left" valign="top">No serious</td><td align="left" valign="top">No serious</td><td align="left" valign="top">Not assessed</td><td align="left" valign="top">High</td></tr><tr><td align="left" valign="top">SLE<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup> classification</td><td align="left" valign="top">17</td><td align="left" valign="top">No serious</td><td align="left" valign="top">No serious</td><td align="left" valign="top">No serious</td><td align="left" valign="top">No serious</td><td align="left" valign="top">Not assessed</td><td align="left" valign="top">High</td></tr><tr><td align="left" valign="top">LN<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup> diagnosis</td><td align="left" valign="top">5</td><td align="left" valign="top">No serious</td><td align="left" valign="top">Some inconsistency (<italic>I</italic>&#x00B2;=56%)</td><td align="left" valign="top">No serious</td><td align="left" valign="top">Imprecision (k=5, max CI width=0.23)</td><td align="left" valign="top">Not assessed</td><td align="left" valign="top">Low</td></tr><tr><td align="left" valign="top">NPSLE<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup> discrimination</td><td align="left" valign="top">7</td><td align="left" valign="top">No serious</td><td align="left" valign="top">No serious</td><td align="left" valign="top">No serious</td><td align="left" valign="top">No serious</td><td align="left" valign="top">Not assessed</td><td align="left" valign="top">High</td></tr><tr><td align="left" valign="top">External validation</td><td align="left" valign="top">9</td><td align="left" valign="top">No serious</td><td align="left" valign="top">No serious</td><td align="left" valign="top">No serious</td><td align="left" valign="top">No serious</td><td align="left" valign="top">Not assessed</td><td align="left" valign="top">High</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>GRADE: Grading of Recommendations Assessment, Development and Evaluation.</p></fn><fn id="table1fn2"><p><sup>b</sup>ML: machine learning.</p></fn><fn id="table1fn3"><p><sup>c</sup>DL: deep learning.</p></fn><fn id="table1fn4"><p><sup>d</sup>SLE: systemic lupus erythematosus.</p></fn><fn id="table1fn5"><p><sup>e</sup>LN: lupus nephritis.</p></fn><fn id="table1fn6"><p><sup>f</sup>NPSLE: neuropsychiatric systemic lupus erythematosus.</p></fn></table-wrap-foot></table-wrap><p>The GRADE framework was used to assess the certainty of evidence for DTA. Certainty starts at high and is rated down for risk of bias when at least one-third of the included studies have at least one QUADAS-AI domain rated as high risk, inconsistency (<italic>I</italic>&#x00B2;&#x003E;50% one level and &#x003E;75% 2 levels), indirectness, imprecision (k&#x003C;5 or 95% CI width&#x003E;0.20 for sensitivity or specificity), and publication bias (Deeks test significance threshold &#x03B1;=.10, where assessed). <italic>I</italic>&#x00B2; represents between-study heterogeneity, and CI denotes confidence interval.</p><p>Five studies (one independent contingency table per study) were included in the primary analysis for LN diagnosis (Task 2) [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref32">32</xref>]. The pooled results showed a sensitivity of 0.93 (95% CI 0.86&#x2010;0.97, 95% PI 0.68&#x2010;0.99) and specificity of 0.95 (95% CI 0.76&#x2010;0.99, 95% PI 0.19&#x2010;1.00). Low-to-moderate between-study heterogeneity (<italic>I</italic>&#x00B2;=18.4% for sensitivity, <italic>I</italic>&#x00B2;=56.4% for specificity) was observed. The wide PI for specificity (0.19&#x2010;1.00), driven by moderate heterogeneity (<italic>I</italic>&#x00B2;=56.4%), indicates variability in reference standard definitions and patient population composition across LN studies. Caution should be exercised when generalizing these estimates to new clinical settings. According to the GRADE assessment, the certainty of evidence was low for LN diagnosis, downgraded for inconsistency (<italic>I</italic>&#x00B2;=56.4% for specificity) and imprecision based on the 5 included studies [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref32">32</xref>] and a maximum CI width of 0.23 (<xref ref-type="table" rid="table1">Table 1</xref>). This low certainty of evidence indicates that the true diagnostic performance may be substantially different from the pooled estimate.</p><p>Seven studies (one independent contingency table per study) were included in the primary analysis for NPSLE discrimination (Task 3) [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref45">45</xref>]. The pooled results showed a sensitivity of 0.88 (95% CI 0.76&#x2010;0.95, 95% PI 0.54&#x2010;0.98) and specificity of 0.89 (95% CI 0.76&#x2010;0.95, 95% PI 0.50&#x2010;0.99). Between-study heterogeneity was low (<italic>I</italic>&#x00B2;=18.0% for sensitivity, <italic>I</italic>&#x00B2;=21.7% for specificity). The low heterogeneity makes the pooled point estimate more clinically interpretable. However, the wide PIs (sensitivity 0.54&#x2010;0.98, specificity 0.50&#x2010;0.99) indicate that in a new clinical setting, the expected performance could vary substantially, reflecting differences in imaging modality, patient selection criteria, and NPSLE definition across studies [<xref ref-type="bibr" rid="ref49">49</xref>]. According to the GRADE assessment, the certainty of evidence was high for NPSLE discrimination (<xref ref-type="table" rid="table1">Table 1</xref>). The forest plots for each clinical task are presented in Figure S4 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p></sec><sec id="s3-5"><title>Exploratory All-Task Pooled Analysis</title><p>An exploratory pooled analysis of all 29 studies (one independent table per study) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref45">45</xref>] yielded a pooled sensitivity of 0.91 (95% CI 0.86&#x2010;0.94, 95% PI 0.56&#x2010;0.99), specificity of 0.94 (95% CI 0.91&#x2010;0.96, 95% PI 0.69&#x2010;0.99), and AUC of 0.969 (SROC), with low between-study heterogeneity (<italic>I</italic>&#x00B2;=23.9% for sensitivity, <italic>I</italic>&#x00B2;=22.9% for specificity; Deeks <italic>P</italic>=.19). This analysis is presented for methodological exploration only. Since these pooled point estimates combined heterogeneous and incomparable clinical tasks, they lacked clinical or statistical significance and should not be used to infer the overall diagnostic performance of ML models for SLE.</p></sec><sec id="s3-6"><title>GRADE Assessment</title><p>The GRADE assessment results are summarized in <xref ref-type="table" rid="table1">Table 1</xref>. The overall certainty of evidence was rated as high for most analyses, including the primary analysis of all 29 studies [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref45">45</xref>], the ML and DL subgroups, SLE classification, NPSLE discrimination, and external validation (<xref ref-type="table" rid="table2">Table 2</xref>). The certainty of evidence for LN diagnosis was rated as low due to inconsistency (<italic>I</italic>&#x00B2;=56.4% for specificity) and imprecision (k=5 studies [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref32">32</xref>], maximum CI width=0.23). These results indicate that while the pooled estimates for most analyses are reliable, the evidence for LN diagnosis should be interpreted with caution, and additional studies are needed to increase confidence in the pooled estimates for this task.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Summary of pooled diagnostic performance across all analyses (29 studies)<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref45">45</xref>].</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Analysis</td><td align="left" valign="bottom">Pooled sensitivity (95% CI)</td><td align="left" valign="bottom">95% PI<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup> (sensitivity)</td><td align="left" valign="bottom">Pooled specificity (95% CI)</td><td align="left" valign="bottom">95% PI (specificity)</td><td align="left" valign="bottom"><italic>I</italic>&#x00B2;<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup> sens (%)</td><td align="left" valign="bottom"><italic>I</italic>&#x00B2; spec (%)</td><td align="left" valign="bottom">Deeks <italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Primary (all 29 studies)</td><td align="left" valign="top">0.91 (0.86&#x2010;0.94)</td><td align="left" valign="top">0.56&#x2010;0.99</td><td align="left" valign="top">0.94 (0.91&#x2010;0.96)</td><td align="left" valign="top">0.69&#x2010;0.99</td><td align="left" valign="top">23.9</td><td align="left" valign="top">22.9</td><td align="left" valign="top">.19</td></tr><tr><td align="left" valign="top">ML<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup> subgroup (n=19)</td><td align="left" valign="top">0.88 (0.81&#x2010;0.93)</td><td align="left" valign="top">0.44&#x2010;0.99</td><td align="left" valign="top">0.94 (0.89&#x2010;0.97)</td><td align="left" valign="top">0.53&#x2010;0.99</td><td align="left" valign="top">27.6</td><td align="left" valign="top">32.7</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup></td></tr><tr><td align="left" valign="top">DL<sup><xref ref-type="table-fn" rid="table2fn6">f</xref></sup> subgroup (n=10)</td><td align="left" valign="top">0.93 (0.91&#x2010;0.95)</td><td align="left" valign="top">0.85&#x2010;0.97</td><td align="left" valign="top">0.95 (0.93&#x2010;0.97)</td><td align="left" valign="top">0.85&#x2010;0.99</td><td align="left" valign="top">5.3</td><td align="left" valign="top">10.0</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">SLE<sup><xref ref-type="table-fn" rid="table2fn7">g</xref></sup> classification (n=17)</td><td align="left" valign="top">0.90 (0.84&#x2010;0.94)</td><td align="left" valign="top">0.53&#x2010;0.99</td><td align="left" valign="top">0.94 (0.92&#x2010;0.96)</td><td align="left" valign="top">0.85&#x2010;0.98</td><td align="left" valign="top">24.7</td><td align="left" valign="top">6.7</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">LN<sup><xref ref-type="table-fn" rid="table2fn8">h</xref></sup> diagnosis (n=5)</td><td align="left" valign="top">0.93 (0.86&#x2010;0.97)</td><td align="left" valign="top">0.68&#x2010;0.99</td><td align="left" valign="top">0.95 (0.76&#x2010;0.99)</td><td align="left" valign="top">0.19&#x2010;1.00</td><td align="left" valign="top">18.4</td><td align="left" valign="top">56.4</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">NPSLE<sup><xref ref-type="table-fn" rid="table2fn9">i</xref></sup> discrimination (n=7)</td><td align="left" valign="top">0.88 (0.76&#x2010;0.95)</td><td align="left" valign="top">0.54&#x2010;0.98</td><td align="left" valign="top">0.89 (0.76&#x2010;0.95)</td><td align="left" valign="top">0.50&#x2010;0.99</td><td align="left" valign="top">18.0</td><td align="left" valign="top">21.7</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">Histopathology (n=7)</td><td align="left" valign="top">0.94 (0.80&#x2010;0.98)</td><td align="left" valign="top">0.28&#x2010;1.00</td><td align="left" valign="top">0.98 (0.92&#x2010;0.99)</td><td align="left" valign="top">0.60&#x2010;1.00</td><td align="left" valign="top">48.2</td><td align="left" valign="top">43.4</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">MRI<sup><xref ref-type="table-fn" rid="table2fn10">j</xref></sup> (n=8)</td><td align="left" valign="top">0.91 (0.81&#x2010;0.96)</td><td align="left" valign="top">0.53&#x2010;0.99</td><td align="left" valign="top">0.88 (0.79&#x2010;0.94)</td><td align="left" valign="top">0.61&#x2010;0.97</td><td align="left" valign="top">24.5</td><td align="left" valign="top">13.9</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">Raman spectroscopy (n=3)</td><td align="left" valign="top">0.97 (0.90&#x2010;0.99)</td><td align="left" valign="top">0.83&#x2010;0.99</td><td align="left" valign="top">0.97 (0.92&#x2010;0.99)</td><td align="left" valign="top">0.88&#x2010;0.99</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">Clinical images (n=5)</td><td align="left" valign="top">0.90 (0.86&#x2010;0.93)</td><td align="left" valign="top">0.79&#x2010;0.95</td><td align="left" valign="top">0.92 (0.88&#x2010;0.95)</td><td align="left" valign="top">0.82&#x2010;0.97</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">External validation (n=9)</td><td align="left" valign="top">0.90 (0.77&#x2010;0.96)</td><td align="left" valign="top">0.32&#x2010;0.99</td><td align="left" valign="top">0.92 (0.83&#x2010;0.96)</td><td align="left" valign="top">0.53&#x2010;0.99</td><td align="left" valign="top">38.2</td><td align="left" valign="top">27.0</td><td align="left" valign="top">&#x2014;</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>PI reflects expected diagnostic performance range in a new similar study. NA, not computed for subgroups with &#x003C;4 studies. Aliyev et al [<xref ref-type="bibr" rid="ref48">48</xref>] and de Araujo et al [<xref ref-type="bibr" rid="ref33">33</xref>] were excluded from all analyses. In the all-table exploratory analysis, studies reporting results for multiple algorithms contribute multiple rows (each representing an independent algorithmic comparison, not double-counting of patients)<italic>.</italic></p></fn><fn id="table2fn2"><p><sup>b</sup>PI: 95% prediction interval.</p></fn><fn id="table2fn3"><p><sup>c</sup><italic>I&#x00B2;</italic>: between-study heterogeneity.</p></fn><fn id="table2fn4"><p><sup>d</sup>ML: machine learning.</p></fn><fn id="table2fn5"><p><sup>e</sup>Not assessed.</p></fn><fn id="table2fn6"><p><sup>f</sup>DL: deep learning.</p></fn><fn id="table2fn7"><p><sup>g</sup>SLE: systemic lupus erythematosus.</p></fn><fn id="table2fn8"><p><sup>h</sup>LN: lupus nephritis.</p></fn><fn id="table2fn9"><p><sup>i</sup>NPSLE: neuropsychiatric systemic lupus erythematosus.</p></fn><fn id="table2fn10"><p><sup>j</sup>MRI: magnetic resonance imaging.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-7"><title>Subgroup and Sensitivity Analyses</title><sec id="s3-7-1"><title>Subgroup Analysis by Validation Strategy</title><p>The model performance was significantly different between internal and external validation. Studies using only internal validation (20 studies) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref25">25</xref>-<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref29">29</xref>-<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref44">44</xref>] showed a pooled sensitivity of 0.92 (95% CI 0.89&#x2010;0.94) and specificity of 0.94 (95% CI 0.91&#x2010;0.96). Studies using independent external validation (9 studies) [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref45">45</xref>] showed a pooled sensitivity of 0.90 (95% CI 0.77&#x2010;0.96, 95% PI 0.32&#x2010;0.99) and specificity of 0.92 (95% CI 0.83&#x2010;0.96, 95% PI 0.53&#x2010;0.99). Nonetheless, only 9 [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref45">45</xref>] of 29 (31%) studies [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref45">45</xref>] performed independent external validation.</p></sec><sec id="s3-7-2"><title>Other Subgroup Analyses</title><p>In the all-table exploratory analysis, non-DL ML algorithms (19 studies with 77 tables) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref21">21</xref>-<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref37">37</xref>-<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref43">43</xref>-<xref ref-type="bibr" rid="ref45">45</xref>] showed a pooled sensitivity of 0.82 (95% CI 0.79&#x2010;0.85) and specificity of 0.89 (95% CI 0.86&#x2010;0.91), with an SROC AUC of 0.917. DL algorithms (10 studies with 47 tables) [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref40">40</xref>-<xref ref-type="bibr" rid="ref42">42</xref>] showed a pooled sensitivity of 0.89 (95% CI 0.87&#x2010;0.91) and specificity of 0.92 (95% CI 0.88&#x2010;0.94), with an SROC AUC of 0.949. In the primary analysis (one table per study), DL models (n=10) [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref40">40</xref>-<xref ref-type="bibr" rid="ref42">42</xref>] achieved a pooled sensitivity of 0.93 (95% CI 0.91&#x2010;0.95, 95% PI 0.85&#x2010;0.97) and a specificity of 0.95 (95% CI 0.93&#x2010;0.97, 95% PI 0.85&#x2010;0.99), compared with traditional ML models (n=19) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref21">21</xref>-<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref37">37</xref>-<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref43">43</xref>-<xref ref-type="bibr" rid="ref45">45</xref>] at sensitivity 0.88 (95% CI 0.81&#x2010;0.93, 95% PI 0.44&#x2010;0.99) and specificity 0.94 (95% CI 0.89&#x2010;0.97, 95% PI 0.53&#x2010;0.99). DL models showed lower heterogeneity (<italic>I</italic>&#x00B2;=5.3% for sensitivity, <italic>I</italic>&#x00B2;=10.0% for specificity) than ML models (<italic>I</italic>&#x00B2;=27.6% for sensitivity, <italic>I</italic>&#x00B2;=32.7% for specificity) and narrower PIs, suggesting more consistent performance across studies. The Cochrane-style forest plots for ML and DL subgroups are shown in <xref ref-type="fig" rid="figure7">Figures 7</xref> and <xref ref-type="fig" rid="figure8">8</xref>, and the subgroup ROC comparison by algorithm type is shown in <xref ref-type="fig" rid="figure9">Figure 9</xref>. The forest plots for ML and DL algorithm subgroups are also presented in Figure S5 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p><fig position="float" id="figure7"><label>Figure 7.</label><caption><p>Cochrane-style forest plot for traditional machine learning models (19 studies): sensitivity and specificity [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref21">21</xref>-<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref37">37</xref>-<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref43">43</xref>-<xref ref-type="bibr" rid="ref45">45</xref>]. Random-effects bivariate model with Knapp-Hartung adjustment; dashed red lines=95% prediction interval. FN: false negative; FP: false positive; ML: machine learning; PI: prediction interval; SENS: sensitivity; SPEC: specificity; TN: true negative; TP: true positive; Wt: study weight.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e90209_fig07.png"/></fig><fig position="float" id="figure8"><label>Figure 8.</label><caption><p>Cochrane-style forest plot for deep learning models (10 studies): sensitivity and specificity [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref40">40</xref>-<xref ref-type="bibr" rid="ref42">42</xref>]. Random-effects bivariate model with Knapp-Hartung adjustment; dashed red lines=95% prediction interval. PI: prediction interval; SENS: sensitivity; SPEC: specificity.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e90209_fig08.png"/></fig><fig position="float" id="figure9"><label>Figure 9.</label><caption><p>Summary receiver operating characteristic curves stratified by algorithm type. (A) Deep learning algorithms (10 studies with 47 tables) [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref40">40</xref>-<xref ref-type="bibr" rid="ref42">42</xref>] and (B) non&#x2013;deep learning machine learning algorithms (19 studies with 77 tables) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref21">21</xref>-<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref37">37</xref>-<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref43">43</xref>-<xref ref-type="bibr" rid="ref45">45</xref>]. AUC: area under the curve; DL: deep learning; ML: machine learning; SENS: sensitivity; SPEC: specificity; SROC: summary receiver operating characteristic.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e90209_fig09.png"/></fig><p>Subgroup analysis by data modality revealed that models using histopathology images (7 studies) [<xref ref-type="bibr" rid="ref20">20</xref>-<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref32">32</xref>] yielded a pooled sensitivity of 0.94 (95% CI 0.80&#x2010;0.98) and specificity of 0.98 (95% CI 0.92&#x2010;0.99). MRI-based models (8 studies) [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref45">45</xref>] yielded a pooled sensitivity of 0.91 (95% CI 0.81&#x2010;0.96) and specificity of 0.88 (95% CI 0.79&#x2010;0.94). Raman spectroscopy&#x2013;based models (3 studies) [<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref41">41</xref>] yielded a pooled sensitivity of 0.97 (95% CI 0.90&#x2010;0.99) and specificity of 0.97 (95% CI 0.92&#x2010;0.99). Clinical image-based models (5 studies) [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref44">44</xref>] yielded a pooled sensitivity of 0.90 (95% CI 0.86&#x2010;0.93) and specificity of 0.92 (95% CI 0.88&#x2010;0.95). No statistically significant differences in performance across modalities were identified in formal subgroup testing. The pooled SROC performance by data modality is presented in Figure S6 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>, and the forest plots for different feature types are presented in Figure S7 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p><p>Subgroup analysis by sample size revealed that studies with &#x2265;100 participants had lower pooled sensitivity (0.89 vs 0.93) and specificity (0.92 vs 0.96) than studies with &#x003C;100 participants, indicating that studies with a small sample size overestimate model performance.</p></sec><sec id="s3-7-3"><title>Sensitivity Analyses</title><p>Leave-one-out analysis showed that no single study had a disproportionate impact on the pooled estimates for any of the 3 primary tasks. Excluding studies with a high overall risk of bias or a small sample size (&#x003C;50 participants) did not significantly change the pooled sensitivity or specificity for the primary analyses, confirming the robustness of the results.</p></sec><sec id="s3-7-4"><title>Small-Study Effects Assessment</title><p>Visual inspection of the funnel plot showed no obvious asymmetry, and the Deeks asymmetry test found no statistically significant small-study effects (<italic>P</italic>=.19 for the primary task-stratified analysis; <italic>P</italic>=.16 for the exploratory all-table analysis; <xref ref-type="fig" rid="figure10">Figure 10</xref>). However, we explicitly note that this analysis has limited statistical power due to the small number of included studies, and cannot rule out researcher-driven optimism bias from selective reporting of prespecified models, threshold tuning, or post hoc model selection, which is a pervasive limitation in ML diagnostic research. The Deeks funnel plots for exploratory analyses are presented in Figure S9 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p><fig position="float" id="figure10"><label>Figure 10.</label><caption><p>Deeks funnel plot asymmetry test for the primary analysis (29 studies) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref45">45</xref>]. <italic>P</italic>=.19, indicating no significant small-study effects. ESS: effective sample size.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e90209_fig10.png"/></fig></sec><sec id="s3-7-5"><title>Comparison Between ML Models and Human Clinicians</title><p>Only 2 studies reported head-to-head comparisons between ML models and human clinicians using the same held-out test dataset, with a total of 6 diagnostic contingency tables. Due to the sparse data, wide CIs, heterogeneous clinical tasks, and varying levels of clinician expertise across studies, no inferential meta-analysis or formal between-group comparison was performed; an exploratory descriptive SROC analysis is presented in Figure S8 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref> for visualization only. In both studies, ML models showed higher overall sensitivity than clinicians, with comparable specificity, but the small number of studies and heterogeneous methods implied that no definitive conclusions can be drawn about the relative diagnostic performance of ML versus human clinicians.</p></sec></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Summary of Principal Findings</title><p>Our principal finding is that ML models show promising in-sample diagnostic accuracy across all 3 prespecified clinical tasks. In the primary task-stratified analyses, the pooled sensitivity and specificity were 0.90 and 0.94, respectively, for SLE classification (n=17) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref29">29</xref>-<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref40">40</xref>-<xref ref-type="bibr" rid="ref44">44</xref>]; 0.93 and 0.95, respectively, for LN diagnosis (n=5) [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref32">32</xref>]; and 0.88 and 0.89, respectively, for NPSLE discrimination (n=7) [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref45">45</xref>], and SROC AUC values of 0.940 (SLE), 0.938 (LN), and 0.877 (NPSLE) in the all-table exploratory analyses. However, these promising results must be interpreted with caution for several reasons. First, all included studies were retrospective, and only 9 [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref45">45</xref>] of 29 (31%) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref45">45</xref>] studies performed independent external validation. Second, while between-study heterogeneity was substantially lower in the updated analysis than the previous version (<italic>I</italic>&#x00B2;=23.9% overall for sensitivity, compared with <italic>I</italic>&#x00B2;=97.5% previously), the PIs remained wide for most analyses, reflecting significant variations in model performance across different clinical populations, imaging modalities, and institutional contexts [<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref50">50</xref>]. Third, reference standards were inconsistent across studies, with some relying on retrospective clinical labels rather than prospectively applied standardized classification criteria [<xref ref-type="bibr" rid="ref51">51</xref>]. Fourth, no included study assessed model calibration, DCA, or net clinical benefit, suggesting that high AUC values do not prove that the model is ready for clinical application [<xref ref-type="bibr" rid="ref52">52</xref>]. Fifth, the GRADE assessment confirmed high certainty of evidence for most analyses but low certainty for LN diagnosis due to inconsistency and imprecision. Hence, additional studies are needed before pooled LN estimates can be confidently applied in clinical practice [<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref47">47</xref>]. Our analysis also confirms that internal validation systematically overestimates ML model performance relative to external validation, which is commonly observed across diagnostic ML research and is not unique to SLE [<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref54">54</xref>].</p></sec><sec id="s4-2"><title>Comparison With Existing Literature and Innovations of This Review</title><p>Our findings build on and address the fundamental limitations of prior systematic reviews in this field, with 4 key contributions to the literature. First, we addressed the core conceptual flaw of prior reviews by performing fully task-stratified meta-analyses, rather than pooling heterogeneous clinical tasks into a single analysis [<xref ref-type="bibr" rid="ref55">55</xref>]. This ensures the clinical interpretability and methodological validity of our pooled estimates and aligns with the core assumptions of DTA meta-analysis. Prior reviews have combined SLE classification, LN activity grading, NPSLE discrimination, and disease activity estimation into a single pooled analysis, yielding results that lack practical value in both statistical and clinical terms. Our task-stratified approach resolves this critical issue [<xref ref-type="bibr" rid="ref56">56</xref>]. Furthermore, this task-stratified approach revealed significant differences in the certainty of evidence across tasks: while SLE classification and NPSLE discrimination achieved high GRADE certainty, LN diagnosis was rated as low certainty, a distinction that would have been obscured by pooling all tasks together.</p><p>Second, we use state-of-the-art statistical methods for DTA meta-analysis, including the HKSJ random-effects model and 95% PIs, which can account for extreme between-study heterogeneity [<xref ref-type="bibr" rid="ref49">49</xref>]. Unlike prior reviews that overinterpreted pooled AUC values in the setting of <italic>I</italic>&#x00B2;&#x003E;95%, we explicitly acknowledged the limitations of pooled point estimates with extreme heterogeneity and used PIs to quantify the real-world variability in model performance [<xref ref-type="bibr" rid="ref57">57</xref>]. For instance, the primary analysis for SLE classification showed a relatively narrow CI (sensitivity 0.84&#x2010;0.94) but a wide PI (sensitivity 0.53&#x2010;0.99). It indicates that while the average performance is precisely estimated, a new study in a different population could yield markedly different results [<xref ref-type="bibr" rid="ref35">35</xref>]. This avoids the statistical misinterpretations in previous reviews that led to inaccurate conclusions [<xref ref-type="bibr" rid="ref58">58</xref>].</p><p>Third, we implemented strict rules for including contingency tables to eliminate data dependency, double counting, and optimism bias. Prior reviews extracted multiple nonindependent contingency tables from the same study (from multiple models, thresholds, or cross-validation splits), treating them as independent observations and leading to artificially narrow CIs and inflated precision [<xref ref-type="bibr" rid="ref59">59</xref>]. In our primary analyses, we adhered to the principle of using only one independent table per study and excluded post hoc best-performing model results, thereby resolving this methodological flaw [<xref ref-type="bibr" rid="ref60">60</xref>]. Fourth, we systematically distinguished between internal and external validation results and quantified the overestimation of performance from internal validation [<xref ref-type="bibr" rid="ref61">61</xref>,<xref ref-type="bibr" rid="ref62">62</xref>]. Prior reviews have confounded internal and external validation results, leading to overestimations of real-world generalizability [<xref ref-type="bibr" rid="ref63">63</xref>]. Internal validation methods, such as k-fold cross-validation and random split-sample validation, estimate model performance within the same dataset used for training, and are therefore susceptible to optimistic bias from dataset-specific patterns, overfitting to demographic or institutional characteristics, and data leakage. In contrast, independent external validation tests the model on entirely new data from different institutions, time periods, or geographic settings, providing a more realistic estimate of real-world performance [<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref62">62</xref>]. In the updated analysis, 9 [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref45">45</xref>] of 29 (31%) [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref34">34</xref>-<xref ref-type="bibr" rid="ref45">45</xref>] studies performed external validation, and the PI for externally validated studies (sensitivity 0.32&#x2010;0.99, specificity 0.53&#x2010;0.99) underscores the high variability of ML model performance across clinical contexts. We prioritized external validation results in all primary analyses [<xref ref-type="bibr" rid="ref64">64</xref>].</p></sec><sec id="s4-3"><title>Limitations of the Included Studies</title><p>The included studies have several major methodological limitations that severely undermine the validity and clinical applicability of their findings. First, all included studies were retrospective in design, which introduces a high risk of bias, including selection bias, data leakage, and overfitting. Retrospective ML model development typically relies on convenience samples that do not reflect the target clinical population, and models may learn spurious correlations from the data that do not generalize to real-world settings [<xref ref-type="bibr" rid="ref65">65</xref>]. No prospective studies of ML for SLE diagnosis were identified in this review. Second, the vast majority of studies used only internal validation, with only 31% (9/29) performing independent external validation. Internal validation is well known to systematically overestimate diagnostic accuracy, particularly in ML studies with flexible modeling pipelines and limited sample sizes [<xref ref-type="bibr" rid="ref53">53</xref>]. Our subgroup analysis confirmed this, showing significantly lower performance in externally validated studies [<xref ref-type="bibr" rid="ref56">56</xref>]. Without independent, multicenter external validation, the real-world generalizability of these models remains unproven [<xref ref-type="bibr" rid="ref66">66</xref>].</p><p>Third, despite the low overall between-study heterogeneity (<italic>I</italic>&#x00B2;=23.9%), the PIs remained wide for most analyses, indicating that the expected performance of ML models in any single new clinical setting could vary substantially. This is partially explained by differences in data modality, validation strategy, and sample size [<xref ref-type="bibr" rid="ref50">50</xref>]. The remaining heterogeneity is likely due to differences in study population, reference standard definition, model development workflow, and preprocessing steps, which limit the generalizability of pooled estimates [<xref ref-type="bibr" rid="ref67">67</xref>]. The wide 95% PIs for most analyses confirm that model performance is highly variable across different clinical settings and populations [<xref ref-type="bibr" rid="ref54">54</xref>]. Fourth, there was significant inconsistency in the reference standards used across studies. Some studies used retrospective clinical labels rather than standardized classification criteria, and a small number of studies used reference standards that were partially informed by the same data used to train the ML model, thereby introducing the risk of circular validation and bias in the reference standard [<xref ref-type="bibr" rid="ref51">51</xref>]. This inconsistency further limits the comparability of results across studies [<xref ref-type="bibr" rid="ref55">55</xref>].</p><p>Fifth, researcher-driven optimism bias is a pervasive risk across all included studies. Many studies reported only results of the best-performing model, without prespecifying the primary model or accounting for multiple comparisons [<xref ref-type="bibr" rid="ref68">68</xref>]. This selective reporting leads to an overestimation of model performance and cannot be ruled out even in the absence of statistically significant small-study effects [<xref ref-type="bibr" rid="ref69">69</xref>]. Finally, none of the included studies assessed the clinical utility of their models. High AUC values alone do not prove that the model can improve patient outcomes in clinical practice; a model with high discriminative accuracy may still cause net harm if it is poorly calibrated, its decision threshold is inappropriate for the target clinical population, or it leads to overdiagnosis, unnecessary invasive testing, or inappropriate immunosuppressive treatment. Model calibration refers to the agreement between the model&#x2019;s predicted probabilities and the actual observed outcome frequencies across the probability range, and is typically assessed using calibration plots, the Hosmer-Lemeshow test, or the expected calibration error (ECE). DCA is the recommended framework for evaluating whether a diagnostic model provides net clinical benefit over default strategies (treat-all or treat-none) across a range of clinically plausible decision thresholds [<xref ref-type="bibr" rid="ref52">52</xref>]. Without such formal assessments and prospective evaluation of the model&#x2019;s impact on diagnostic workflow, patient management decisions, and clinical outcomes, it is impossible to determine whether these ML models would provide meaningful benefit to patients with SLE in real-world clinical practice.</p></sec><sec id="s4-4"><title>Limitations of This Systematic Review</title><p>This review has several limitations that should be acknowledged. First, we only included studies published in English, which may introduce language bias [<xref ref-type="bibr" rid="ref70">70</xref>]. Second, we excluded conference abstracts, which may lead to publication bias, as negative or nonsignificant results are less likely to be published in full-text peer-reviewed journals. Third, the number of included studies for some subgroup analyses was small, limiting statistical power, and these analyses are therefore presented as exploratory only. Fourth, we were unable to perform meta-analysis of individual patient data, which would have allowed for more robust adjustment for confounding factors and more detailed subgroup analyses [<xref ref-type="bibr" rid="ref71">71</xref>]. Finally, we were unable to assess the risk of data leakage in the included studies, as this is often not reported in detail, which may lead to an overestimation of model performance in the original studies [<xref ref-type="bibr" rid="ref72">72</xref>,<xref ref-type="bibr" rid="ref73">73</xref>]. Additionally, the GRADE framework, while widely used for assessing certainty of evidence, has limitations when applied to DTA studies. It was originally developed for intervention studies and may not fully capture all sources of uncertainty in diagnostic accuracy meta-analyses [<xref ref-type="bibr" rid="ref46">46</xref>].</p></sec><sec id="s4-5"><title>Implications for Future Research and Clinical Practice</title><p>The findings of this review have critical implications for future research and clinical practice. For future research, we make the following evidence-based recommendations. First, future ML studies for SLE diagnosis must be prospectively designed, with preregistration of the study protocol, model development plan, and statistical analysis plan, to reduce the risk of bias and selective reporting [<xref ref-type="bibr" rid="ref65">65</xref>,<xref ref-type="bibr" rid="ref68">68</xref>]. Second, all ML models must undergo independent, multicenter external validation in cohorts that reflect the target clinical population, to confirm real-world generalizability [<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref62">62</xref>,<xref ref-type="bibr" rid="ref66">66</xref>]. Third, future studies must use clearly defined, clinically homogeneous diagnostic tasks and consistent reference standards to reduce heterogeneity and enable meaningful comparison between models [<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref69">69</xref>]. Fourth, all future studies must assess clinical utility, including model calibration, DCA to quantify net clinical benefit, and evaluation of the model&#x2019;s impact on diagnostic workflow and patient outcomes [<xref ref-type="bibr" rid="ref52">52</xref>]. Fifth, future studies should perform prespecified, head-to-head comparisons between ML models and board-certified rheumatologists, using standardized test datasets and clearly defined clinician expertise levels, to determine the incremental value of ML-assisted diagnosis over standard clinical care.</p><p>For clinical practice, our review confirms that ML models for SLE diagnosis are not yet ready for routine clinical application [<xref ref-type="bibr" rid="ref74">74</xref>]. The current evidence is of variable quality, with GRADE certainty ranging from high (for most analyses) to low (for LN diagnosis), due to exclusively retrospective designs, lack of independent external validation, wide PIs, and absence of clinical utility assessment [<xref ref-type="bibr" rid="ref75">75</xref>]. There is currently no evidence that these models improve patient outcomes in real-world clinical settings [<xref ref-type="bibr" rid="ref76">76</xref>]. At this stage, ML should only be used as an auxiliary tool to support expert clinical judgment, in the context of prospective research studies, until robust, externally validated, clinically beneficial models are developed [<xref ref-type="bibr" rid="ref77">77</xref>,<xref ref-type="bibr" rid="ref78">78</xref>].</p></sec><sec id="s4-6"><title>Conclusion</title><p>This is the first methodologically rigorous, task-stratified systematic review and meta-analysis of ML models for SLE diagnosis, with formal GRADE assessment of certainty of evidence, addressing the core conceptual and statistical flaws of prior syntheses. ML models show promising in-sample diagnostic accuracy for 3 distinct SLE-related clinical tasks, but the current evidence is of variable certainty (high for most analyses, low for LN diagnosis), limited by pervasive methodological weaknesses, including exclusively retrospective designs, lack of independent external validation, wide PIs despite low <italic>I</italic>&#x00B2;, and absence of clinical utility assessment. The distinction between the narrow CIs (reflecting precision of the average effect) and the wide PIs (reflecting expected variability across new settings) is critical for clinical interpretation: the pooled estimates should not be taken as guaranteed performance in any individual new setting [<xref ref-type="bibr" rid="ref35">35</xref>]. Future research must prioritize prospectively registered, multicenter, externally validated studies with standardized clinical tasks, harmonized reference standards, and formal assessment of clinical utility, to support the safe and effective translation of ML tools into clinical care for SLE.</p></sec></sec></body><back><ack><p>The authors declare the use of generative AI (GenAI) in the research and writing process. According to the GAIDeT (Generative AI Delegation Taxonomy; 2025), the following tasks were delegated to GenAI tools under full human supervision: proofreading and editing. The GenAI tool used was ChatGPT-4o and multiple ChatGPT-5 versions. Responsibility for the final manuscript lies entirely with the authors. GenAI tools are not listed as authors and do not bear responsibility for the final outcomes. The declaration was submitted under collective responsibility.</p><p>The AI was used solely for language polishing and grammatical correction. All generated suggestions were reviewed and approved by the authors. GenAI tools (ChatGPT 4o, OpenAI) were used to a limited extent during the preparation of this manuscript, solely for minor language polishing of individual sentences and formatting of reference lists, in accordance with JMIR Publications guidelines. GenAI was not used at any stage of study design, literature search, study selection, data extraction, risk of bias assessment, statistical analysis, data interpretation, or formulation of scientific conclusions. All statistical analyses were independently performed by the authors using R (version 4.4.2) with the mada package. All figures were generated by the authors using R. The GRADE assessment, QUADAS-AI evaluation, and all clinical interpretations were performed entirely by the authors without AI assistance. The final content of the manuscript, including all scientific interpretation, statistical analysis, and clinical conclusions, was fully reviewed, edited, and approved by all authors, who take full responsibility for the accuracy and integrity of the work.</p></ack><notes><sec><title>Funding</title><p>The authors declare that no funds, grants, or other financial support were received during the preparation of this manuscript.</p></sec><sec><title>Data Availability</title><p>All data generated or analyzed during this study are included in this published article and its supplementary information files.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: BW, ZW, YL, FG</p><p>Formal analysis: YL, FG</p><p>Investigation: YL, FG</p><p>Methodology: BW</p><p>Supervision: FG</p><p>Writing &#x2013; original draft: BW</p><p>Writing &#x2013; review &#x0026; editing: ZW, YL, FG</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ACR</term><def><p>American College of Rheumatology</p></def></def-item><def-item><term id="abb2">AUC</term><def><p>area under the curve</p></def></def-item><def-item><term id="abb3">CT</term><def><p>computed tomography</p></def></def-item><def-item><term id="abb4">DCA</term><def><p>decision-curve analysis</p></def></def-item><def-item><term id="abb5">DL</term><def><p>deep learning</p></def></def-item><def-item><term id="abb6">DTA</term><def><p>diagnostic test accuracy</p></def></def-item><def-item><term id="abb7">ECE</term><def><p>expected calibration error</p></def></def-item><def-item><term id="abb8">EEG</term><def><p>electroencephalogram</p></def></def-item><def-item><term id="abb9">EHR</term><def><p>electronic health record</p></def></def-item><def-item><term id="abb10">EULAR</term><def><p>European League Against Rheumatism</p></def></def-item><def-item><term id="abb11">FN</term><def><p>false negative</p></def></def-item><def-item><term id="abb12">FODI</term><def><p>finger optical diffusion imaging</p></def></def-item><def-item><term id="abb13">FP</term><def><p>false positive</p></def></def-item><def-item><term id="abb14">FTIR</term><def><p>Fourier transform infrared spectroscopy</p></def></def-item><def-item><term id="abb15">GRADE</term><def><p>Grading of Recommendations Assessment, Development and Evaluation</p></def></def-item><def-item><term id="abb16">HKSJ</term><def><p>Hartung-Knapp-Sidik-Jonkman</p></def></def-item><def-item><term id="abb17">HSROC</term><def><p>hierarchical summary receiver operating characteristic</p></def></def-item><def-item><term id="abb18">ISN</term><def><p>International Society of Nephrology</p></def></def-item><def-item><term id="abb19">LN</term><def><p>lupus nephritis</p></def></def-item><def-item><term id="abb20">ML</term><def><p>machine learning</p></def></def-item><def-item><term id="abb21">MRI</term><def><p>magnetic resonance imaging</p></def></def-item><def-item><term id="abb22">NPSLE</term><def><p>neuropsychiatric systemic lupus erythematosus</p></def></def-item><def-item><term id="abb23">NPV</term><def><p>negative predictive value</p></def></def-item><def-item><term id="abb24">OCT</term><def><p>optical coherence tomography</p></def></def-item><def-item><term id="abb25">PI</term><def><p>prediction interval</p></def></def-item><def-item><term id="abb26">PPV</term><def><p>positive predictive value</p></def></def-item><def-item><term id="abb27">PRISMA</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses</p></def></def-item><def-item><term id="abb28">PRISMA-DTA</term><def><p>PRISMA for Diagnostic Test Accuracy</p></def></def-item><def-item><term id="abb29">PRISMA-S</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses literature search extension</p></def></def-item><def-item><term id="abb30">PROSPERO</term><def><p>International Prospective Register of Systematic Reviews</p></def></def-item><def-item><term id="abb31">QUADAS-AI</term><def><p>Quality Assessment of Diagnostic Accuracy Studies for Artificial Intelligence</p></def></def-item><def-item><term id="abb32">ROC</term><def><p>receiver operating characteristic</p></def></def-item><def-item><term id="abb33">RPS</term><def><p>Renal Pathology Society</p></def></def-item><def-item><term id="abb34">SLE</term><def><p>systemic lupus erythematosus</p></def></def-item><def-item><term id="abb35">SLICC</term><def><p>Systemic Lupus Collaborating Clinics</p></def></def-item><def-item><term id="abb36">SROC</term><def><p>summary receiver operating characteristic</p></def></def-item><def-item><term id="abb37">TN</term><def><p>true negative</p></def></def-item><def-item><term id="abb38">TP</term><def><p>true positive</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mills</surname><given-names>JA</given-names> </name></person-group><article-title>Systemic lupus erythematosus</article-title><source>N Engl J Med</source><year>1994</year><month>06</month><day>30</day><volume>330</volume><issue>26</issue><fpage>1871</fpage><lpage>1879</lpage><pub-id pub-id-type="doi">10.1056/NEJM199406303302608</pub-id><pub-id pub-id-type="medline">8196732</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cervera</surname><given-names>R</given-names> </name><name name-style="western"><surname>Rodr&#x00ED;guez-Pint&#x00F3;</surname><given-names>I</given-names> </name><name name-style="western"><surname>Espinosa</surname><given-names>G</given-names> </name></person-group><article-title>The diagnosis and clinical management of the catastrophic antiphospholipid syndrome: a comprehensive review</article-title><source>J Autoimmun</source><year>2018</year><month>08</month><volume>92</volume><fpage>1</fpage><lpage>11</lpage><pub-id pub-id-type="doi">10.1016/j.jaut.2018.05.007</pub-id><pub-id pub-id-type="medline">29779928</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Durcan</surname><given-names>L</given-names> </name><name name-style="western"><surname>O&#x2019;Dwyer</surname><given-names>T</given-names> </name><name name-style="western"><surname>Petri</surname><given-names>M</given-names> </name></person-group><article-title>Management strategies and future directions for systemic lupus erythematosus in adults</article-title><source>Lancet</source><year>2019</year><month>06</month><day>8</day><volume>393</volume><issue>10188</issue><fpage>2332</fpage><lpage>2343</lpage><pub-id pub-id-type="doi">10.1016/S0140-6736(19)30237-5</pub-id><pub-id pub-id-type="medline">31180030</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kiriakidou</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ching</surname><given-names>CL</given-names> </name></person-group><article-title>Systemic lupus erythematosus</article-title><source>Ann Intern Med</source><year>2020</year><month>06</month><day>2</day><volume>172</volume><issue>11</issue><fpage>ITC81</fpage><lpage>ITC96</lpage><pub-id pub-id-type="doi">10.7326/AITC202006020</pub-id><pub-id pub-id-type="medline">32479157</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nandakumar</surname><given-names>KS</given-names> </name><name name-style="western"><surname>N&#x00FC;ndel</surname><given-names>K</given-names> </name></person-group><article-title>Editorial: Systemic lupus erythematosus - predisposition factors, pathogenesis, diagnosis, treatment and disease models</article-title><source>Front Immunol</source><year>2022</year><volume>13</volume><fpage>1118180</fpage><pub-id pub-id-type="doi">10.3389/fimmu.2022.1118180</pub-id><pub-id pub-id-type="medline">36591294</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zieve</surname><given-names>GW</given-names> </name><name name-style="western"><surname>Khusial</surname><given-names>PR</given-names> </name></person-group><article-title>The anti-Sm immune response in autoimmunity and cell biology</article-title><source>Autoimmun Rev</source><year>2003</year><month>09</month><volume>2</volume><issue>5</issue><fpage>235</fpage><lpage>240</lpage><pub-id pub-id-type="doi">10.1016/s1568-9972(03)00018-1</pub-id><pub-id pub-id-type="medline">12965173</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><article-title>Antiphospholipid syndrome</article-title><source>Nat Rev Dis Primers</source><year>2018</year><month>01</month><day>11</day><volume>4</volume><issue>1</issue><fpage>17104</fpage><pub-id pub-id-type="doi">10.1038/nrdp.2017.104</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Adamichou</surname><given-names>C</given-names> </name><name name-style="western"><surname>Nikolopoulos</surname><given-names>D</given-names> </name><name name-style="western"><surname>Genitsaridi</surname><given-names>I</given-names> </name><etal/></person-group><article-title>In an early SLE cohort the ACR-1997, SLICC-2012 and EULAR/ACR-2019 criteria classify non-overlapping groups of patients: use of all three criteria ensures optimal capture for clinical studies while their modification earlier classification and treatment</article-title><source>Ann Rheum Dis</source><year>2020</year><month>02</month><volume>79</volume><issue>2</issue><fpage>232</fpage><lpage>241</lpage><pub-id pub-id-type="doi">10.1136/annrheumdis-2019-216155</pub-id><pub-id pub-id-type="medline">31704720</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Han</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Fucosylation of anti-dsDNA IgG1 correlates with disease activity of treatment-na&#x00EF;ve systemic lupus erythematosus patients</article-title><source>EBioMedicine</source><year>2022</year><month>03</month><volume>77</volume><fpage>103883</fpage><pub-id pub-id-type="doi">10.1016/j.ebiom.2022.103883</pub-id><pub-id pub-id-type="medline">35182998</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Barbhaiya</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zuily</surname><given-names>S</given-names> </name><name name-style="western"><surname>Naden</surname><given-names>R</given-names> </name><etal/></person-group><article-title>2023 ACR/EULAR antiphospholipid syndrome classification criteria</article-title><source>Ann Rheum Dis</source><year>2023</year><month>10</month><volume>82</volume><issue>10</issue><fpage>1258</fpage><lpage>1270</lpage><pub-id pub-id-type="doi">10.1136/ard-2023-224609</pub-id><pub-id pub-id-type="medline">37640450</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Appenzeller</surname><given-names>S</given-names> </name><name name-style="western"><surname>Pereira</surname><given-names>DR</given-names> </name><name name-style="western"><surname>Julio</surname><given-names>PR</given-names> </name><name name-style="western"><surname>Reis</surname><given-names>F</given-names> </name><name name-style="western"><surname>Rittner</surname><given-names>L</given-names> </name><name name-style="western"><surname>Marini</surname><given-names>R</given-names> </name></person-group><article-title>Neuropsychiatric manifestations in childhood-onset systemic lupus erythematosus</article-title><source>Lancet Child Adolesc Health</source><year>2022</year><month>08</month><volume>6</volume><issue>8</issue><fpage>571</fpage><lpage>581</lpage><pub-id pub-id-type="doi">10.1016/S2352-4642(22)00157-2</pub-id><pub-id pub-id-type="medline">35841921</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fangerau</surname><given-names>H</given-names> </name></person-group><article-title>Artifical intelligence in surgery: ethical considerations in the light of social trends in the perception of health and medicine</article-title><source>EFORT Open Rev</source><year>2024</year><month>05</month><day>10</day><volume>9</volume><issue>5</issue><fpage>323</fpage><lpage>328</lpage><pub-id pub-id-type="doi">10.1530/EOR-24-0029</pub-id><pub-id pub-id-type="medline">38726973</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Handelman</surname><given-names>GS</given-names> </name><name name-style="western"><surname>Kok</surname><given-names>HK</given-names> </name><name name-style="western"><surname>Chandra</surname><given-names>RV</given-names> </name><name name-style="western"><surname>Razavi</surname><given-names>AH</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Asadi</surname><given-names>H</given-names> </name></person-group><article-title>eDoctor: machine learning and the future of medicine</article-title><source>J Intern Med</source><year>2018</year><month>12</month><volume>284</volume><issue>6</issue><fpage>603</fpage><lpage>619</lpage><pub-id pub-id-type="doi">10.1111/joim.12822</pub-id><pub-id pub-id-type="medline">30102808</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hong</surname><given-names>G</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Artificial intelligence-based model for lymph node metastases detection on whole slide images in bladder cancer: a retrospective, multicentre, diagnostic study</article-title><source>Lancet Oncol</source><year>2023</year><month>04</month><volume>24</volume><issue>4</issue><fpage>360</fpage><lpage>370</lpage><pub-id pub-id-type="doi">10.1016/S1470-2045(23)00061-X</pub-id><pub-id pub-id-type="medline">36893772</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McInnes</surname><given-names>MDF</given-names> </name><name name-style="western"><surname>Moher</surname><given-names>D</given-names> </name><name name-style="western"><surname>Thombs</surname><given-names>BD</given-names> </name><etal/></person-group><article-title>Preferred Reporting Items for a Systematic Review and Meta-analysis of Diagnostic Test Accuracy Studies: the PRISMA-DTA statement</article-title><source>JAMA</source><year>2018</year><month>01</month><day>23</day><volume>319</volume><issue>4</issue><fpage>388</fpage><lpage>396</lpage><pub-id pub-id-type="doi">10.1001/jama.2017.19163</pub-id><pub-id pub-id-type="medline">29362800</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yuan</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Quan</surname><given-names>T</given-names> </name><name name-style="western"><surname>Song</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Guan</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>T</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>R</given-names> </name></person-group><article-title>Noise-immune extreme ensemble learning for early diagnosis of neuropsychiatric systemic lupus erythematosus</article-title><source>IEEE J Biomed Health Inform</source><year>2022</year><month>07</month><volume>26</volume><issue>7</issue><fpage>3495</fpage><lpage>3506</lpage><pub-id pub-id-type="doi">10.1109/JBHI.2022.3164937</pub-id><pub-id pub-id-type="medline">35380977</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Luo</surname><given-names>X</given-names> </name><name name-style="western"><surname>Piao</surname><given-names>S</given-names> </name><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Multi-lesion radiomics model for discrimination of relapsing-remitting multiple sclerosis and neuropsychiatric systemic lupus erythematosus</article-title><source>Eur Radiol</source><year>2022</year><month>08</month><volume>32</volume><issue>8</issue><fpage>5700</fpage><lpage>5710</lpage><pub-id pub-id-type="doi">10.1007/s00330-022-08653-2</pub-id><pub-id pub-id-type="medline">35243524</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Cao</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Artificial intelligence: a powerful paradigm for scientific research</article-title><source>Innovation (Camb)</source><year>2021</year><month>11</month><day>28</day><volume>2</volume><issue>4</issue><fpage>100179</fpage><pub-id pub-id-type="doi">10.1016/j.xinn.2021.100179</pub-id><pub-id pub-id-type="medline">34877560</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Marone</surname><given-names>A</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Evaluation of SLE arthritis using frequency domain optical imaging</article-title><source>Lupus Sci Med</source><year>2021</year><month>08</month><volume>8</volume><issue>1</issue><fpage>e000495</fpage><pub-id pub-id-type="doi">10.1136/lupus-2021-000495</pub-id><pub-id pub-id-type="medline">34462335</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zheng</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Ding</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Deep learning-based artificial intelligence system for automatic assessment of glomerular pathological findings in lupus nephritis</article-title><source>Diagnostics (Basel)</source><year>2021</year><month>10</month><day>26</day><volume>11</volume><issue>11</issue><fpage>1983</fpage><pub-id pub-id-type="doi">10.3390/diagnostics11111983</pub-id><pub-id pub-id-type="medline">34829330</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jorge</surname><given-names>A</given-names> </name><name name-style="western"><surname>Castro</surname><given-names>VM</given-names> </name><name name-style="western"><surname>Barnado</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Identifying lupus patients in electronic health records: development and validation of machine learning algorithms and application of rule-based algorithms</article-title><source>Semin Arthritis Rheum</source><year>2019</year><month>08</month><volume>49</volume><issue>1</issue><fpage>84</fpage><lpage>90</lpage><pub-id pub-id-type="doi">10.1016/j.semarthrit.2019.01.002</pub-id><pub-id pub-id-type="medline">30665626</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>WD</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>SN</given-names> </name><etal/></person-group><article-title>Lupus nephritis or not? A simple and clinically friendly machine learning pipeline to help diagnosis of lupus nephritis</article-title><source>Inflamm Res</source><year>2023</year><month>06</month><volume>72</volume><issue>6</issue><fpage>1315</fpage><lpage>1324</lpage><pub-id pub-id-type="doi">10.1007/s00011-023-01755-7</pub-id><pub-id pub-id-type="medline">37300586</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Adamichou</surname><given-names>C</given-names> </name><name name-style="western"><surname>Genitsaridi</surname><given-names>I</given-names> </name><name name-style="western"><surname>Nikolopoulos</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Lupus or not? SLE Risk Probability Index (SLERPI): a simple, clinician-friendly machine learning-based model to assist the diagnosis of systemic lupus erythematosus</article-title><source>Ann Rheum Dis</source><year>2021</year><month>06</month><volume>80</volume><issue>6</issue><fpage>758</fpage><lpage>766</lpage><pub-id pub-id-type="doi">10.1136/annrheumdis-2020-219069</pub-id><pub-id pub-id-type="medline">33568388</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Du</surname><given-names>J</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Pang</surname><given-names>L</given-names> </name><etal/></person-group><article-title>A machine learning model for identifying systemic lupus erythematosus through laboratory information system and electronic medical record</article-title><source>Clin Exp Rheumatol</source><year>2024</year><month>03</month><volume>42</volume><issue>3</issue><fpage>702</fpage><lpage>712</lpage><pub-id pub-id-type="doi">10.55563/clinexprheumatol/jvdrpc</pub-id><pub-id pub-id-type="medline">37976115</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Inglese</surname><given-names>F</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>M</given-names> </name><name name-style="western"><surname>Steup-Beekman</surname><given-names>GM</given-names> </name><etal/></person-group><article-title>MRI-based classification of neuropsychiatric systemic lupus erythematosus patients with self-supervised contrastive learning</article-title><source>Front Neurosci</source><year>2022</year><volume>16</volume><fpage>695888</fpage><pub-id pub-id-type="doi">10.3389/fnins.2022.695888</pub-id><pub-id pub-id-type="medline">35250439</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Qin</surname><given-names>X</given-names> </name><name name-style="western"><surname>Xia</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Noninvasive evaluation of lupus nephritis activity using a radiomics machine learning model based on ultrasound</article-title><source>J Inflamm Res</source><year>2023</year><volume>16</volume><fpage>433</fpage><lpage>441</lpage><pub-id pub-id-type="doi">10.2147/JIR.S398399</pub-id><pub-id pub-id-type="medline">36761904</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Samundeswari</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ramalinga</surname><given-names>V</given-names> </name><name name-style="western"><surname>Latha</surname><given-names>B</given-names> </name><name name-style="western"><surname>Palanivel</surname><given-names>S</given-names> </name></person-group><article-title>Pattern classification techniques for the classification of cutaneous manifestations of systemic lupus erythematosus</article-title><source>Pak J Biotechnol</source><year>2018</year><month>06</month><access-date>2025-01-12</access-date><volume>15</volume><issue>2</issue><fpage>333</fpage><lpage>337</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://pjbt.org/index.php/pjbt/article/view/400">https://pjbt.org/index.php/pjbt/article/view/400</ext-link></comment></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Simos</surname><given-names>NJ</given-names> </name><name name-style="western"><surname>Dimitriadis</surname><given-names>SI</given-names> </name><name name-style="western"><surname>Kavroulakis</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Quantitative identification of functional connectivity disturbances in neuropsychiatric lupus based on resting-state fMRI: a robust machine learning approach</article-title><source>Brain Sci</source><year>2020</year><month>10</month><day>25</day><volume>10</volume><issue>11</issue><fpage>777</fpage><pub-id pub-id-type="doi">10.3390/brainsci10110777</pub-id><pub-id pub-id-type="medline">33113768</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alves</surname><given-names>P</given-names> </name><name name-style="western"><surname>Bandaria</surname><given-names>J</given-names> </name><name name-style="western"><surname>Leavy</surname><given-names>MB</given-names> </name><etal/></person-group><article-title>Validation of a machine learning approach to estimate Systemic Lupus Erythematosus Disease Activity Index score categories and application in a real-world dataset</article-title><source>RMD Open</source><year>2021</year><month>05</month><volume>7</volume><issue>2</issue><fpage>e001586</fpage><pub-id pub-id-type="doi">10.1136/rmdopen-2021-001586</pub-id><pub-id pub-id-type="medline">34016712</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Dey</surname><given-names>SB</given-names> </name><name name-style="western"><surname>Mansoor</surname><given-names>N</given-names> </name></person-group><article-title>A butterfly malar rash detection model for early systemic lupus erythematosus diagnosis</article-title><conf-name>2023 26th International Conference on Computer and Information Technology (ICCIT)</conf-name><conf-date>Dec 13-15, 2023</conf-date><conf-loc>Cox&#x2019;s Bazar, Bangladesh</conf-loc><fpage>1</fpage><lpage>6</lpage><pub-id pub-id-type="doi">10.1109/ICCIT60459.2023.10441593</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>S</given-names> </name><name name-style="western"><surname>Masood</surname><given-names>A</given-names> </name><name name-style="western"><surname>Li</surname><given-names>T</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>G</given-names> </name><name name-style="western"><surname>Dai</surname><given-names>R</given-names> </name></person-group><article-title>Deep learning-enabled automatic screening of SLE diseases and LR using OCT images</article-title><source>Vis Comput</source><year>2023</year><month>08</month><volume>39</volume><issue>8</issue><fpage>3259</fpage><lpage>3269</lpage><pub-id pub-id-type="doi">10.1007/s00371-023-02945-4</pub-id><pub-id pub-id-type="medline">37361461</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>M</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Lupus nephritis diagnosis using enhanced moth flame algorithm with support vector machines</article-title><source>Comput Biol Med</source><year>2022</year><month>06</month><volume>145</volume><fpage>105435</fpage><pub-id pub-id-type="doi">10.1016/j.compbiomed.2022.105435</pub-id><pub-id pub-id-type="medline">35397339</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Silaide de Ara&#x00FA;jo J&#x00FA;nior</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sato</surname><given-names>EI</given-names> </name><name name-style="western"><surname>Silva de Souza</surname><given-names>AW</given-names> </name><etal/></person-group><article-title>Development of an instrument to predict proliferative histological class in lupus nephritis based on clinical and laboratory data</article-title><source>Lupus</source><year>2023</year><month>02</month><volume>32</volume><issue>2</issue><fpage>216</fpage><lpage>224</lpage><pub-id pub-id-type="doi">10.1177/09612033221143933</pub-id><pub-id pub-id-type="medline">36461171</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Rapid diagnosis of systemic lupus erythematosus by Raman spectroscopy combined with spiking neural network</article-title><source>Spectrochim Acta A Mol Biomol Spectrosc</source><year>2024</year><month>04</month><day>5</day><volume>310</volume><fpage>123904</fpage><pub-id pub-id-type="doi">10.1016/j.saa.2024.123904</pub-id><pub-id pub-id-type="medline">38262298</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>C</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>C</given-names> </name><etal/></person-group><article-title>SLE diagnosis research based on SERS combined with a multi-modal fusion method</article-title><source>Spectrochim Acta A Mol Biomol Spectrosc</source><year>2024</year><month>07</month><volume>315</volume><fpage>124296</fpage><pub-id pub-id-type="doi">10.1016/j.saa.2024.124296</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>U</surname><given-names>P</given-names> </name><name name-style="western"><surname>M</surname><given-names>S</given-names> </name><name name-style="western"><surname>E</surname><given-names>SS</given-names> </name><name name-style="western"><surname>P</surname><given-names>S</given-names> </name><name name-style="western"><surname>J</surname><given-names>C</given-names> </name></person-group><article-title>Systemic lupus erythematosus detection using deep learning with auxiliary parameters</article-title><conf-name>2023 Second International Conference on Electrical, Electronics, Information and Communication Technologies (ICEEICT)</conf-name><conf-date>Apr 5-7, 2023</conf-date><conf-loc>Trichirappalli, India</conf-loc><fpage>1</fpage><lpage>7</lpage><pub-id pub-id-type="doi">10.1109/ICEEICT56924.2023.10157562</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>WD</given-names> </name><name name-style="western"><surname>Qin</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Systemic lupus erythematosus with high disease activity identification based on machine learning</article-title><source>Inflamm Res</source><year>2023</year><month>09</month><volume>72</volume><issue>9</issue><fpage>1909</fpage><lpage>1918</lpage><pub-id pub-id-type="doi">10.1007/s00011-023-01793-1</pub-id><pub-id pub-id-type="medline">37725103</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Simos</surname><given-names>NJ</given-names> </name><name name-style="western"><surname>Manikis</surname><given-names>GC</given-names> </name><name name-style="western"><surname>Papadaki</surname><given-names>E</given-names> </name><name name-style="western"><surname>Kavroulakis</surname><given-names>E</given-names> </name><name name-style="western"><surname>Bertsias</surname><given-names>G</given-names> </name><name name-style="western"><surname>Marias</surname><given-names>K</given-names> </name></person-group><article-title>Machine learning classification of neuropsychiatric systemic lupus erythematosus patients using resting-state fmri functional connectivity</article-title><conf-name>2019 IEEE International Conference on Imaging Systems and Techniques (IST)</conf-name><conf-date>Dec 9-10, 2019</conf-date><conf-loc>Abu Dhabi, United Arab Emirates</conf-loc><fpage>1</fpage><lpage>6</lpage><pub-id pub-id-type="doi">10.1109/IST48021.2019.9010078</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tan</surname><given-names>G</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Cui</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Dou</surname><given-names>H</given-names> </name><name name-style="western"><surname>Zheng</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>T</given-names> </name></person-group><article-title>A noise-immune reinforcement learning method for early diagnosis of neuropsychiatric systemic lupus erythematosus</article-title><source>Math Biosci Eng</source><year>2022</year><month>01</month><day>4</day><volume>19</volume><issue>3</issue><fpage>2219</fpage><lpage>2239</lpage><pub-id pub-id-type="doi">10.3934/mbe.2022104</pub-id><pub-id pub-id-type="medline">35240783</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Human-multimodal deep learning collaboration in &#x201C;precise&#x201D; diagnosis of lupus erythematosus subtypes and similar skin diseases</article-title><source>J Eur Acad Dermatol Venereol</source><year>2024</year><month>12</month><volume>38</volume><issue>12</issue><fpage>2268</fpage><lpage>2279</lpage><pub-id pub-id-type="doi">10.1111/jdv.20031</pub-id><pub-id pub-id-type="medline">38619440</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhou</surname><given-names>X</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>C</given-names> </name><name name-style="western"><surname>Lv</surname><given-names>X</given-names> </name><etal/></person-group><article-title>CMACF: Transformer-based cross-modal attention cross-fusion model for systemic lupus erythematosus diagnosis combining Raman spectroscopy, FTIR spectroscopy, and metabolomics</article-title><source>Inf Process Manag</source><year>2024</year><month>11</month><volume>61</volume><issue>6</issue><fpage>103804</fpage><pub-id pub-id-type="doi">10.1016/j.ipm.2024.103804</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>T</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>S</given-names> </name><name name-style="western"><surname>Guan</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>A deep learning system for detecting systemic lupus erythematosus from retinal images</article-title><source>Cell Rep Med</source><year>2025</year><month>07</month><day>15</day><volume>6</volume><issue>7</issue><fpage>102203</fpage><pub-id pub-id-type="doi">10.1016/j.xcrm.2025.102203</pub-id><pub-id pub-id-type="medline">40570853</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bosnal&#x0131;</surname><given-names>B</given-names> </name><name name-style="western"><surname>T&#x00FC;rk</surname><given-names>E</given-names> </name><name name-style="western"><surname>&#x00D6;&#x011F;&#x00FC;t</surname><given-names>TS</given-names> </name><etal/></person-group><article-title>Effectiveness of artificial intelligence in classification of connective tissue diseases in patients with anti-nuclear antibody (ANA) positivity</article-title><source>Comput Biol Chem</source><year>2026</year><month>02</month><volume>120</volume><issue>Pt 1</issue><fpage>108679</fpage><pub-id pub-id-type="doi">10.1016/j.compbiolchem.2025.108679</pub-id><pub-id pub-id-type="medline">40945129</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Jian</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Machine learning-based multiclass model for autoimmune disease diagnosis and classification through nailfold videocapillaroscopy features</article-title><source>RMD Open</source><year>2026</year><month>03</month><day>4</day><volume>12</volume><issue>1</issue><fpage>e006393</fpage><pub-id pub-id-type="doi">10.1136/rmdopen-2025-006393</pub-id><pub-id pub-id-type="medline">41781159</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><name name-style="western"><surname>Tian</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Free water in the hippocampal cingulum as a Radiomic biomarker for Identifying inflammatory neuropsychiatric Lupus: a cross-sectional case-control study</article-title><source>J Autoimmun</source><year>2026</year><month>05</month><volume>160</volume><fpage>103560</fpage><pub-id pub-id-type="doi">10.1016/j.jaut.2026.103560</pub-id><pub-id pub-id-type="medline">41974094</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sch&#x00FC;nemann</surname><given-names>HJ</given-names> </name><name name-style="western"><surname>Mustafa</surname><given-names>RA</given-names> </name><name name-style="western"><surname>Brozek</surname><given-names>J</given-names> </name><etal/></person-group><article-title>GRADE guidelines: 21 part 1. Study design, risk of bias, and indirectness in rating the certainty across a body of evidence for test accuracy</article-title><source>J Clin Epidemiol</source><year>2020</year><month>06</month><volume>122</volume><fpage>129</fpage><lpage>141</lpage><pub-id pub-id-type="doi">10.1016/j.jclinepi.2019.12.020</pub-id><pub-id pub-id-type="medline">32060007</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sch&#x00FC;nemann</surname><given-names>HJ</given-names> </name><name name-style="western"><surname>Mustafa</surname><given-names>RA</given-names> </name><name name-style="western"><surname>Brozek</surname><given-names>J</given-names> </name><etal/></person-group><article-title>GRADE guidelines: 21 part 2. Test accuracy: inconsistency, imprecision, publication bias, and other domains for rating the certainty of evidence and presenting it in evidence profiles and summary of findings tables</article-title><source>J Clin Epidemiol</source><year>2020</year><month>06</month><volume>122</volume><fpage>142</fpage><lpage>152</lpage><pub-id pub-id-type="doi">10.1016/j.jclinepi.2019.12.021</pub-id><pub-id pub-id-type="medline">32058069</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aliyev</surname><given-names>E</given-names> </name><name name-style="western"><surname>Ugur</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Cam</surname><given-names>V</given-names> </name><etal/></person-group><article-title>Closed circuit artificial &#x0131;ntelligence model named morgaf for childhood onset systemic lupus erythematosus diagnosis</article-title><source>Sci Rep</source><year>2025</year><month>07</month><day>1</day><volume>15</volume><issue>1</issue><fpage>20868</fpage><pub-id pub-id-type="doi">10.1038/s41598-025-92964-z</pub-id><pub-id pub-id-type="medline">40595005</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Varoquaux</surname><given-names>G</given-names> </name><name name-style="western"><surname>Cheplygina</surname><given-names>V</given-names> </name></person-group><article-title>Machine learning for medical imaging: methodological failures and recommendations for the future</article-title><source>NPJ Digit Med</source><year>2022</year><month>04</month><day>12</day><volume>5</volume><issue>1</issue><fpage>48</fpage><pub-id pub-id-type="doi">10.1038/s41746-022-00592-y</pub-id><pub-id pub-id-type="medline">35413988</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Torab-Miandoab</surname><given-names>A</given-names> </name><name name-style="western"><surname>Samad-Soltani</surname><given-names>T</given-names> </name><name name-style="western"><surname>Jodati</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rezaei-Hachesu</surname><given-names>P</given-names> </name></person-group><article-title>Interoperability of heterogeneous health information systems: a systematic literature review</article-title><source>BMC Med Inform Decis Mak</source><year>2023</year><month>01</month><day>24</day><volume>23</volume><issue>1</issue><fpage>18</fpage><pub-id pub-id-type="doi">10.1186/s12911-023-02115-5</pub-id><pub-id pub-id-type="medline">36694161</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Beam</surname><given-names>AL</given-names> </name><name name-style="western"><surname>Manrai</surname><given-names>AK</given-names> </name><name name-style="western"><surname>Ghassemi</surname><given-names>M</given-names> </name></person-group><article-title>Challenges to the reproducibility of machine learning models in health care</article-title><source>JAMA</source><year>2020</year><month>01</month><day>28</day><volume>323</volume><issue>4</issue><fpage>305</fpage><lpage>306</lpage><pub-id pub-id-type="doi">10.1001/jama.2019.20866</pub-id><pub-id pub-id-type="medline">31904799</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hofer</surname><given-names>IS</given-names> </name><name name-style="western"><surname>Burns</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kendale</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wanderer</surname><given-names>JP</given-names> </name></person-group><article-title>Realistically integrating machine learning into clinical practice: a road map of opportunities, challenges, and a potential future</article-title><source>Anesth Analg</source><year>2020</year><month>05</month><volume>130</volume><issue>5</issue><fpage>1115</fpage><lpage>1118</lpage><pub-id pub-id-type="doi">10.1213/ANE.0000000000004575</pub-id><pub-id pub-id-type="medline">32287118</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Goodacre</surname><given-names>R</given-names> </name></person-group><article-title>On splitting training and validation set: a comparative study of cross-validation, bootstrap and systematic sampling for estimating the generalization performance of supervised learning</article-title><source>J Anal Test</source><year>2018</year><volume>2</volume><issue>3</issue><fpage>249</fpage><lpage>262</lpage><pub-id pub-id-type="doi">10.1007/s41664-018-0068-2</pub-id><pub-id pub-id-type="medline">30842888</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cabitza</surname><given-names>F</given-names> </name><name name-style="western"><surname>Campagner</surname><given-names>A</given-names> </name><name name-style="western"><surname>Soares</surname><given-names>F</given-names> </name><etal/></person-group><article-title>The importance of being external. methodological insights for the external validation of machine learning models in medicine</article-title><source>Comput Methods Programs Biomed</source><year>2021</year><month>09</month><volume>208</volume><fpage>106288</fpage><pub-id pub-id-type="doi">10.1016/j.cmpb.2021.106288</pub-id><pub-id pub-id-type="medline">34352688</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ahsan</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Luna</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Siddique</surname><given-names>Z</given-names> </name></person-group><article-title>Machine-learning-based disease diagnosis: a comprehensive review</article-title><source>Health Care (Don Mills)</source><volume>10</volume><issue>3</issue><fpage>541</fpage><pub-id pub-id-type="doi">10.3390/healthcare10030541</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xue</surname><given-names>P</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Qin</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Deep learning in image-based breast and cervical cancer detection: a systematic review and meta-analysis</article-title><source>NPJ Digit Med</source><year>2022</year><month>02</month><day>15</day><volume>5</volume><issue>1</issue><fpage>19</fpage><pub-id pub-id-type="doi">10.1038/s41746-022-00559-z</pub-id><pub-id pub-id-type="medline">35169217</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Arbet</surname><given-names>J</given-names> </name><name name-style="western"><surname>Brokamp</surname><given-names>C</given-names> </name><name name-style="western"><surname>Meinzen-Derr</surname><given-names>J</given-names> </name><name name-style="western"><surname>Trinkley</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Spratt</surname><given-names>HM</given-names> </name></person-group><article-title>Lessons and tips for designing a machine learning study using EHR data</article-title><source>J Clin Transl Sci</source><year>2020</year><month>07</month><day>24</day><volume>5</volume><issue>1</issue><fpage>e21</fpage><pub-id pub-id-type="doi">10.1017/cts.2020.513</pub-id><pub-id pub-id-type="medline">33948244</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kiani</surname><given-names>AK</given-names> </name><name name-style="western"><surname>Naureen</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Pheby</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Methodology for clinical research</article-title><source>J Prev Med Hyg</source><year>2022</year><month>06</month><volume>63</volume><issue>2 Suppl 3</issue><fpage>E267</fpage><lpage>E278</lpage><pub-id pub-id-type="doi">10.15167/2421-4248/jpmh2022.63.2S3.2769</pub-id><pub-id pub-id-type="medline">36479476</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kohli</surname><given-names>MD</given-names> </name><name name-style="western"><surname>Summers</surname><given-names>RM</given-names> </name><name name-style="western"><surname>Geis</surname><given-names>JR</given-names> </name></person-group><article-title>Medical image data and datasets in the era of machine learning-whitepaper from the 2016 C-MIMI Meeting Dataset Session</article-title><source>J Digit Imaging</source><year>2017</year><month>08</month><volume>30</volume><issue>4</issue><fpage>392</fpage><lpage>399</lpage><pub-id pub-id-type="doi">10.1007/s10278-017-9976-3</pub-id><pub-id pub-id-type="medline">28516233</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Karimi</surname><given-names>D</given-names> </name><name name-style="western"><surname>Dou</surname><given-names>H</given-names> </name><name name-style="western"><surname>Warfield</surname><given-names>SK</given-names> </name><name name-style="western"><surname>Gholipour</surname><given-names>A</given-names> </name></person-group><article-title>Deep learning with noisy labels: exploring techniques and remedies in medical image analysis</article-title><source>Med Image Anal</source><year>2020</year><month>10</month><volume>65</volume><fpage>101759</fpage><pub-id pub-id-type="doi">10.1016/j.media.2020.101759</pub-id><pub-id pub-id-type="medline">32623277</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Grannis</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Vest</surname><given-names>JR</given-names> </name><etal/></person-group><article-title>Evaluating the effect of data standardization and validation on patient matching accuracy</article-title><source>J Am Med Inform Assoc</source><year>2019</year><month>05</month><day>1</day><volume>26</volume><issue>5</issue><fpage>447</fpage><lpage>456</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocy191</pub-id><pub-id pub-id-type="medline">30848796</pub-id></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aggarwal</surname><given-names>R</given-names> </name><name name-style="western"><surname>Sounderajah</surname><given-names>V</given-names> </name><name name-style="western"><surname>Martin</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Diagnostic accuracy of deep learning in medical imaging: a systematic review and meta-analysis</article-title><source>NPJ Digit Med</source><year>2021</year><month>04</month><day>7</day><volume>4</volume><issue>1</issue><fpage>65</fpage><pub-id pub-id-type="doi">10.1038/s41746-021-00438-z</pub-id><pub-id pub-id-type="medline">33828217</pub-id></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tian</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Yao</surname><given-names>X</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>Q</given-names> </name></person-group><article-title>Global epidemiology of systemic lupus erythematosus: a comprehensive systematic analysis and modelling study</article-title><source>Ann Rheum Dis</source><year>2023</year><month>03</month><volume>82</volume><issue>3</issue><fpage>351</fpage><lpage>356</lpage><pub-id pub-id-type="doi">10.1136/ard-2022-223035</pub-id><pub-id pub-id-type="medline">36241363</pub-id></nlm-citation></ref><ref id="ref64"><label>64</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Giuffr&#x00E8;</surname><given-names>M</given-names> </name><name name-style="western"><surname>Shung</surname><given-names>DL</given-names> </name></person-group><article-title>Harnessing the power of synthetic data in healthcare: innovation, application, and privacy</article-title><source>NPJ Digit Med</source><year>2023</year><month>10</month><day>9</day><volume>6</volume><issue>1</issue><fpage>186</fpage><pub-id pub-id-type="doi">10.1038/s41746-023-00927-3</pub-id><pub-id pub-id-type="medline">37813960</pub-id></nlm-citation></ref><ref id="ref65"><label>65</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ueda</surname><given-names>D</given-names> </name><name name-style="western"><surname>Yamamoto</surname><given-names>A</given-names> </name><name name-style="western"><surname>Takashima</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Training, validation, and test of deep learning models for classification of receptor expressions in breast cancers from mammograms</article-title><source>JCO Precis Oncol</source><year>2021</year><month>11</month><volume>5</volume><fpage>543</fpage><lpage>551</lpage><pub-id pub-id-type="doi">10.1200/PO.20.00176</pub-id><pub-id pub-id-type="medline">34994603</pub-id></nlm-citation></ref><ref id="ref66"><label>66</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kelly</surname><given-names>CJ</given-names> </name><name name-style="western"><surname>Karthikesalingam</surname><given-names>A</given-names> </name><name name-style="western"><surname>Suleyman</surname><given-names>M</given-names> </name><name name-style="western"><surname>Corrado</surname><given-names>G</given-names> </name><name name-style="western"><surname>King</surname><given-names>D</given-names> </name></person-group><article-title>Key challenges for delivering clinical impact with artificial intelligence</article-title><source>BMC Med</source><year>2019</year><month>10</month><day>29</day><volume>17</volume><issue>1</issue><fpage>195</fpage><pub-id pub-id-type="doi">10.1186/s12916-019-1426-2</pub-id><pub-id pub-id-type="medline">31665002</pub-id></nlm-citation></ref><ref id="ref67"><label>67</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alowais</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Alghamdi</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Alsuhebany</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Revolutionizing healthcare: the role of artificial intelligence in clinical practice</article-title><source>BMC Med Educ</source><year>2023</year><month>09</month><day>22</day><volume>23</volume><issue>1</issue><fpage>689</fpage><pub-id pub-id-type="doi">10.1186/s12909-023-04698-z</pub-id><pub-id pub-id-type="medline">37740191</pub-id></nlm-citation></ref><ref id="ref68"><label>68</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rajkomar</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hardt</surname><given-names>M</given-names> </name><name name-style="western"><surname>Howell</surname><given-names>MD</given-names> </name><name name-style="western"><surname>Corrado</surname><given-names>G</given-names> </name><name name-style="western"><surname>Chin</surname><given-names>MH</given-names> </name></person-group><article-title>Ensuring fairness in machine learning to advance health equity</article-title><source>Ann Intern Med</source><year>2018</year><month>12</month><day>18</day><volume>169</volume><issue>12</issue><fpage>866</fpage><lpage>872</lpage><pub-id pub-id-type="doi">10.7326/M18-1990</pub-id><pub-id pub-id-type="medline">30508424</pub-id></nlm-citation></ref><ref id="ref69"><label>69</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aringer</surname><given-names>M</given-names> </name><name name-style="western"><surname>Costenbader</surname><given-names>K</given-names> </name><name name-style="western"><surname>Daikh</surname><given-names>D</given-names> </name><etal/></person-group><article-title>2019 European League Against Rheumatism/American College of Rheumatology classification criteria for systemic lupus erythematosus</article-title><source>Arthritis Rheumatol</source><year>2019</year><month>09</month><volume>71</volume><issue>9</issue><fpage>1400</fpage><lpage>1412</lpage><pub-id pub-id-type="doi">10.1002/art.40930</pub-id><pub-id pub-id-type="medline">31385462</pub-id></nlm-citation></ref><ref id="ref70"><label>70</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>R</given-names> </name><name name-style="western"><surname>Desai</surname><given-names>NR</given-names> </name><name name-style="western"><surname>Ross</surname><given-names>JS</given-names> </name><etal/></person-group><article-title>Publication and reporting of clinical trial results: cross sectional analysis across academic medical centers</article-title><source>BMJ</source><year>2016</year><month>02</month><day>17</day><volume>352</volume><fpage>i637</fpage><pub-id pub-id-type="doi">10.1136/bmj.i637</pub-id><pub-id pub-id-type="medline">26888209</pub-id></nlm-citation></ref><ref id="ref71"><label>71</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Monaghan</surname><given-names>TF</given-names> </name><name name-style="western"><surname>Rahman</surname><given-names>SN</given-names> </name><name name-style="western"><surname>Agudelo</surname><given-names>CW</given-names> </name><etal/></person-group><article-title>Foundational statistical principles in medical research: sensitivity, specificity, positive predictive value, and negative predictive value</article-title><source>Medicina (B Aires)</source><year>2021</year><month>05</month><day>16</day><volume>57</volume><issue>5</issue><fpage>503</fpage><pub-id pub-id-type="doi">10.3390/medicina57050503</pub-id></nlm-citation></ref><ref id="ref72"><label>72</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aydin</surname><given-names>OU</given-names> </name><name name-style="western"><surname>Taha</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Hilbert</surname><given-names>A</given-names> </name><etal/></person-group><article-title>An evaluation of performance measures for arterial brain vessel segmentation</article-title><source>BMC Med Imaging</source><year>2021</year><month>07</month><day>16</day><volume>21</volume><issue>1</issue><fpage>113</fpage><pub-id pub-id-type="doi">10.1186/s12880-021-00644-x</pub-id><pub-id pub-id-type="medline">34271876</pub-id></nlm-citation></ref><ref id="ref73"><label>73</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chicco</surname><given-names>D</given-names> </name><name name-style="western"><surname>T&#x00F6;tsch</surname><given-names>N</given-names> </name><name name-style="western"><surname>Jurman</surname><given-names>G</given-names> </name></person-group><article-title>The Matthews correlation coefficient (MCC) is more reliable than balanced accuracy, bookmaker informedness, and markedness in two-class confusion matrix evaluation</article-title><source>BioData Min</source><year>2021</year><month>02</month><day>4</day><volume>14</volume><issue>1</issue><fpage>13</fpage><pub-id pub-id-type="doi">10.1186/s13040-021-00244-z</pub-id><pub-id pub-id-type="medline">33541410</pub-id></nlm-citation></ref><ref id="ref74"><label>74</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Di Matteo</surname><given-names>A</given-names> </name><name name-style="western"><surname>Smerilli</surname><given-names>G</given-names> </name><name name-style="western"><surname>Cipolletta</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Imaging of joint and soft tissue involvement in systemic lupus erythematosus</article-title><source>Curr Rheumatol Rep</source><year>2021</year><month>07</month><day>16</day><volume>23</volume><issue>9</issue><fpage>73</fpage><pub-id pub-id-type="doi">10.1007/s11926-021-01040-8</pub-id><pub-id pub-id-type="medline">34269905</pub-id></nlm-citation></ref><ref id="ref75"><label>75</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Komura</surname><given-names>D</given-names> </name><name name-style="western"><surname>Ishikawa</surname><given-names>S</given-names> </name></person-group><article-title>Machine learning methods for histopathological image analysis</article-title><source>Comput Struct Biotechnol J</source><year>2018</year><volume>16</volume><fpage>34</fpage><lpage>42</lpage><pub-id pub-id-type="doi">10.1016/j.csbj.2018.01.001</pub-id><pub-id pub-id-type="medline">30275936</pub-id></nlm-citation></ref><ref id="ref76"><label>76</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hanna</surname><given-names>MG</given-names> </name><name name-style="western"><surname>Ardon</surname><given-names>O</given-names> </name><name name-style="western"><surname>Reuter</surname><given-names>VE</given-names> </name><etal/></person-group><article-title>Integrating digital pathology into clinical practice</article-title><source>Mod Pathol</source><year>2022</year><month>02</month><volume>35</volume><issue>2</issue><fpage>152</fpage><lpage>164</lpage><pub-id pub-id-type="doi">10.1038/s41379-021-00929-0</pub-id><pub-id pub-id-type="medline">34599281</pub-id></nlm-citation></ref><ref id="ref77"><label>77</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Price</surname><given-names>WN</given-names>  <suffix>II</suffix></name><name name-style="western"><surname>Cohen</surname><given-names>IG</given-names> </name></person-group><article-title>Privacy in the age of medical big data</article-title><source>Nat Med</source><year>2019</year><month>01</month><volume>25</volume><issue>1</issue><fpage>37</fpage><lpage>43</lpage><pub-id pub-id-type="doi">10.1038/s41591-018-0272-7</pub-id><pub-id pub-id-type="medline">30617331</pub-id></nlm-citation></ref><ref id="ref78"><label>78</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lu</surname><given-names>P</given-names> </name><name name-style="western"><surname>Oetjen</surname><given-names>KA</given-names> </name><name name-style="western"><surname>Bender</surname><given-names>DE</given-names> </name><etal/></person-group><article-title>IMC-Denoise: a content aware denoising pipeline to enhance imaging mass cytometry</article-title><source>Nat Commun</source><year>2023</year><month>03</month><day>23</day><volume>14</volume><issue>1</issue><fpage>1601</fpage><pub-id pub-id-type="doi">10.1038/s41467-023-37123-6</pub-id><pub-id pub-id-type="medline">36959190</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Deviations from PROSPERO preregistered protocol (CRD42024545109).</p><media xlink:href="jmir_v28i1e90209_app1.xlsx" xlink:title="XLSX File, 11 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Full database search strategies (Cochrane-Compliant).</p><media xlink:href="jmir_v28i1e90209_app2.docx" xlink:title="DOCX File, 15 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Study characteristics, model development and validation, diagnostic performance, subgroup analyses, human-clinician comparisons, and publication bias.</p><media xlink:href="jmir_v28i1e90209_app3.docx" xlink:title="DOCX File, 4945 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Grading of Recommendations Assessment, Development, and Evaluation summary-of-findings tables for the diagnostic accuracy of machine learning and deep learning models for systemic lupus erythematosus&#x2013;related diagnostic tasks.</p><media xlink:href="jmir_v28i1e90209_app4.docx" xlink:title="DOCX File, 27 KB"/></supplementary-material><supplementary-material id="app5"><label>Checklist 1</label><p>PRISMA 2020 checklist.</p><media xlink:href="jmir_v28i1e90209_app5.docx" xlink:title="DOCX File, 29 KB"/></supplementary-material><supplementary-material id="app6"><label>Checklist 2</label><p>PRISMA-DTA complete compliance checklist.</p><media xlink:href="jmir_v28i1e90209_app6.docx" xlink:title="DOCX File, 14 KB"/></supplementary-material><supplementary-material id="app7"><label>Checklist 3</label><p>PRISMA-S complete compliance checklist.</p><media xlink:href="jmir_v28i1e90209_app7.docx" xlink:title="DOCX File, 13 KB"/></supplementary-material><supplementary-material id="app8"><label>Checklist 4</label><p>PRISMA 2020 complete compliance checklist.</p><media xlink:href="jmir_v28i1e90209_app8.docx" xlink:title="DOCX File, 16 KB"/></supplementary-material></app-group></back></article>