<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="review-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e93618</article-id><article-id pub-id-type="doi">10.2196/93618</article-id><article-categories><subj-group subj-group-type="heading"><subject>Review</subject></subj-group></article-categories><title-group><article-title>Cognitive Workload and Mental Burden in Health Care Professionals Interacting With AI: Systematic Review and Meta-Analysis</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Gong</surname><given-names>Eun Jeong</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Bang</surname><given-names>Chang Seok</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Lee</surname><given-names>Jae Jun</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff4">4</xref></contrib></contrib-group><aff id="aff1"><institution>Institute for Liver and Digestive Diseases, Hallym University</institution><addr-line>Chuncheon</addr-line><country>Republic of Korea</country></aff><aff id="aff2"><institution>Institute of New Frontier Research, Hallym University College of Medicine</institution><addr-line>Chuncheon</addr-line><country>Republic of Korea</country></aff><aff id="aff3"><institution>Department of Internal Medicine, Hallym University College of Medicine</institution><addr-line>Sakju-ro 77</addr-line><addr-line>Chuncheon</addr-line><country>Republic of Korea</country></aff><aff id="aff4"><institution>Department of Anesthesiology and Pain Medicine, Hallym University College of Medicine</institution><addr-line>Chuncheon</addr-line><country>Republic of Korea</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Brini</surname><given-names>Stefano</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Rose</surname><given-names>Danielle</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Yin</surname><given-names>Rong</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Chang Seok Bang, MD, PhD, Department of Internal Medicine, Hallym University College of Medicine, Sakju-ro 77, Chuncheon, 24253, Republic of Korea, 82 82 33 240 582; <email>csbang@hallym.ac.kr</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>4</day><month>8</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e93618</elocation-id><history><date date-type="received"><day>16</day><month>02</month><year>2026</year></date><date date-type="rev-recd"><day>14</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>15</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Eun Jeong Gong, Chang Seok Bang, Jae Jun Lee. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 4.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e93618"/><abstract><sec><title>Background</title><p>AI adoption in health care has accelerated rapidly, with ambient documentation tools, diagnostic imaging AI, and clinical decision support systems (CDSSs) entering routine practice. However, the cognitive demands placed on clinicians supervising these systems remain understudied. Specifically, the concept of verification burden requires closer examination. Consequently, institutional decision-makers lack a structured, certainty-graded evidence base regarding the true impact of AI on clinician workload and burnout.</p></sec><sec><title>Objective</title><p>This study aimed to systematically review evidence on cognitive workload and burnout in health care professionals that use AI-powered clinical tools, quantify pooled effects under a conservative inferential framework, and assess certainty of evidence by AI category.</p></sec><sec sec-type="methods"><title>Methods</title><p>The study was registered in PROSPERO (CRD420261284298) and reported per PRISMA 2020 and PRISMA-S guidelines. We searched MEDLINE, Embase, Web of Science, and Cochrane CENTRAL (January 2015-2026) for studies measuring cognitive workload or burnout using validated instruments (NASA Task Load Index [NASA-TLX] and Professional Fulfillment Index [PFI]) among health care professionals using clinical AI. Risk of bias was assessed using ROB 2.0 and ROBINS-I; certainty was rated using GRADE. Meta-analyses applied Hartung-Knapp-Sidik-Jonkman adjustment with restricted maximum likelihood estimation, incorporating prediction intervals (PIs).</p></sec><sec sec-type="results"><title>Results</title><p>We included 21 studies representing 2885 health care professionals across 7 countries. The synthesis demonstrated that the cognitive impact of clinical AI varies according to its specific application. Pooled analyses of ambient AI documentation showed statistically significant reductions in NASA-TLX temporal demand (SMD &#x2212;1.46, 95% CI &#x2212;2.81 to &#x2212;0.11; k=2; <italic>I</italic><sup>2</sup>=31.1%) and effort (SMD &#x2212;1.29, 95% CI &#x2212;2.16 to &#x2212;0.42; k=2; <italic>I</italic><sup>2</sup>=0%), PFI work exhaustion (MD &#x2212;0.35, 95% CI &#x2212;0.58 to &#x2212;0.12; k=3; <italic>I</italic><sup>2</sup>=0%; 95% PI &#x2212;1.03 to 0.33), and burnout prevalence (OR 0.47, 95% CI 0.25-0.86; k=3; <italic>I</italic><sup>2</sup>=0%; 95% PI 0.06-3.82). Two pools favored ambient AI but did not reach significance at k=2: NASA-TLX mental demand (SMD &#x2212;1.29, 95% CI &#x2212;3.64 to 1.07) and documentation time (SMD &#x2212;0.24, 95% CI &#x2212;1.10 to 0.61). Diagnostic imaging AI and CDSS showed mixed or paradoxically increased workload. GRADE certainty was moderate for cognitive workload reduction with ambient AI, low for burnout reduction with ambient AI, and very low for imaging AI and CDSS outcomes.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>This review combines validated workload instruments, meta-analysis, and PIs in health care AI, delivering a GRADE certainty assessment across 5 AI categories that prior accuracy- or efficiency-focused reviews have not provided. Ambient AI documentation was associated with reduced cognitive workload and burnout, but only in voluntary early-adopter cohorts and based on few studies; the conservative CIs were wide and, where estimable, PIs crossed the null. Findings inform institutional pilots with prospective workload measurement, regulatory human-factors evaluation of AI medical devices, and human-centered AI design. Net benefit on the health care workforce remains an open empirical question.</p></sec><sec><title>Trial Registration</title><p>PROSPERO CRD420261284298; <ext-link ext-link-type="uri" xlink:href="https://www.crd.york.ac.uk/PROSPERO/view/CRD420261284298">https://www.crd.york.ac.uk/PROSPERO/view/CRD420261284298</ext-link></p></sec></abstract><kwd-group><kwd>artificial intelligence</kwd><kwd>cognitive workload</kwd><kwd>mental fatigue</kwd><kwd>human-AI interaction</kwd><kwd>alert fatigue</kwd><kwd>verification burden</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>The rapid adoption of artificial intelligence (AI) in health care is changing clinical workflows across various medical specialties. From diagnostic imaging analysis to ambient documentation scribes, AI-powered tools are increasingly integrated into routine clinical practice. According to the American Medical Association, physician AI use nearly doubled from 38% in 2023 to 66% in 2024, reflecting rapid integration compared to historical health care technology adoptions [<xref ref-type="bibr" rid="ref1">1</xref>].</p><p>These tools are intended to alleviate the administrative burden that has been identified as a primary driver of clinician burnout. This clinician crisis peaked at 62.8% prevalence in 2021, though rates have since declined to approximately 45% as of 2023 [<xref ref-type="bibr" rid="ref2">2</xref>]. Algorithmic solutions are considered practical largely because they can automate cognitively demanding tasks, particularly clinical documentation. A landmark time-motion study demonstrated that physicians spend 49.2% of their office day on electronic health record and desk work combined, with only 27% on direct clinical face time, and an additional 1&#x2010;2 hours of electronic health record work each evening [<xref ref-type="bibr" rid="ref3">3</xref>]. Early evidence from ambient AI scribes suggests meaningful reductions in documentation time, and these tools are frequently promoted to reduce administrative burden [<xref ref-type="bibr" rid="ref4">4</xref>]. However, it remains unclear whether these systems reduce overall cognitive burden or shift it from content generation to verification.</p><p>Recent commercial deployments in 2024 and 2025 have accelerated this enthusiasm. Tierney et al [<xref ref-type="bibr" rid="ref5">5</xref>] reported substantial reductions in documentation burden following an enterprise-wide rollout of an ambient AI scribe to over 3000 clinicians, and Albrecht et al [<xref ref-type="bibr" rid="ref6">6</xref>] described similar quality-improvement gains in a multispecialty implementation. These 2 enterprise deployments illustrate the pace of commercial scaling but do not contribute outcome data to this review because neither used a validated cognitive-workload instrument, such as the National Aeronautics and Space Administration Task Load Index (NASA-TLX), or the Physician Task Load Index, or a validated burnout instrument as a prespecified primary outcome. Consequently, neither met the eligibility criteria (detailed in the Methods section). Throughout this text, commercial deployment data, theoretical frameworks, and prior reviews provide background only; the primary, secondary, and exploratory outcomes reported in the Results section derive solely from the 21 studies that met the eligibility criteria.</p><p>Subsequent trials and observational studies reported reductions in NASA-TLX subscales, Professional Fulfillment Index (PFI) work exhaustion, and burnout prevalence among adopters of ambient AI documentation systems [<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref10">10</xref>]. However, this rapidly growing literature is concentrated on a small number of commercial products (predominantly Abridge and Dragon Ambient Experience [DAX] Copilot) deployed in voluntary early-adopter cohorts, mostly in US outpatient primary care, with follow-up periods rarely exceeding 3 months. The rapid pace of commercial deployment has outpaced the accumulation of independent, multiproduct, long-term human-factors evidence. This disparity motivates a systematic quantitative synthesis of existing data to map current empirical gaps.</p><p>Integrating AI into clinical workflows modifies the clinician&#x2019;s role from an active creator of clinical content to a supervisor of AI-generated outputs. This transition imposes novel cognitive demands that differ qualitatively from traditional documentation tasks. Clinicians must engage in continuous verification of AI recommendations with clinical judgment, a phenomenon characterized as verification burden. Human factors research suggests that such monitoring tasks are cognitively demanding in ways that human cognitive architecture is not well suited to sustain, particularly given the ease with which grammatically polished AI-generated text may bypass critical evaluation [<xref ref-type="bibr" rid="ref11">11</xref>]. Conceptually, verification burden differs from extraneous cognitive load, which is a construct from cognitive load theory describing task-irrelevant demands imposed by suboptimal interface or instructional design. Instead, verification burden is inherent to the AI-supervision task itself, aligning closely with Bainbridge&#x2019;s &#x201C;ironies of automation&#x201D; framework [<xref ref-type="bibr" rid="ref12">12</xref>], where automation generates monitoring demands that may exceed the cognitive savings provided.</p><p>Theoretical frameworks from automation science provide important context for understanding this relationship. Bainbridge&#x2019;s [<xref ref-type="bibr" rid="ref12">12</xref>] work on the ironies of automation described how systems designed to reduce human workload often create new cognitive demands through the requirement for vigilance and oversight. The out-of-the-loop performance problem further suggests that operators monitoring automated systems experience degraded situation awareness and diminished capacity to detect errors [<xref ref-type="bibr" rid="ref13">13</xref>]. Applied to health care AI, these frameworks predict that clinicians may experience automation complacency, defined as a reduced tendency to verify AI outputs, leading to missed errors despite sustained cognitive effort. These constructs, including verification burden, automation bias, and automation complacency, were not measured by validated instruments in any included study; they are framed throughout as a hypothesis-generating interpretive lens [<xref ref-type="bibr" rid="ref14">14</xref>].</p><p>Despite the proliferation of health care AI tools and concern about their human factors implications, empirical evidence on cognitive workload remains sparse. A 2024 systematic review examining AI implementation in medical imaging found only 3 studies addressing clinician workload, noting that no study assessed workload separately in terms of cognitive workload changes, and describing this gap as remarkable [<xref ref-type="bibr" rid="ref15">15</xref>]. Similarly, while numerous studies have examined AI&#x2019;s impact on documentation time, few have used validated instruments to measure the subjective cognitive experience of clinicians interacting with these systems. Only a small subset of ambient AI scribe studies uses validated cognitive workload instruments such as the NASA-TLX; most rely instead on documentation time, single-item satisfaction ratings, or burnout scales as proxies. Studies of diagnostic imaging AI and alert-based clinical decision support systems (CDSSs) have likewise not consistently measured cognitive experience. For instance, a prospective evaluation of computer-aided detection (CADe) for prostate magnetic resonance imaging (MRI) found no reduction in radiologist workload despite improved diagnostic performance [<xref ref-type="bibr" rid="ref16">16</xref>], and a usability evaluation of a pediatric sepsis prediction model documented increased perceived workload and alert fatigue rather than the expected reduction [<xref ref-type="bibr" rid="ref17">17</xref>].</p><p>The heterogeneity of AI applications, outcome measures, and study designs has limited previous attempts at a quantitative synthesis with explicit certainty ratings. Consequently, institutional decision-makers and regulatory authorities lack a structured evidence base to guide implementation. This gap is significant for patient safety and clinician well-being, given that cognitive workload is a predictor of medical errors, burnout, and workforce attrition [<xref ref-type="bibr" rid="ref18">18</xref>]. If AI tools reduce physical documentation effort while imposing equivalent or greater cognitive demands through verification requirements, the net benefit for clinicians may be minimal or even negative. Understanding this trade-off is essential for evidence-based AI implementation and the design of human-centered clinical AI systems. This systematic review aims to synthesize the available evidence on cognitive workload and mental burden experienced by health care professionals when interacting with AI-powered clinical tools. Specifically, we sought (1) to quantify cognitive workload associated with AI-assisted clinical tasks using validated instruments, (2) to identify factors that influence cognitive burden across different AI applications and clinical domains, and (3) to evaluate related constructs including burnout, automation bias, and verification burden that may contribute to clinicians&#x2019; psychological strain when working with AI systems.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Protocol and Registration</title><p>This systematic review and meta-analysis was conducted in accordance with the PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) 2020 expanded checklist [<xref ref-type="bibr" rid="ref19">19</xref>] and the PRISMA-S (Preferred Reporting Items for Systematic Reviews and Meta-Analyses Literature Search Extension) extension for search reporting [<xref ref-type="bibr" rid="ref20">20</xref>]. Completed reporting checklists are provided as <xref ref-type="supplementary-material" rid="app2">Checklist 1</xref> (PRISMA 2020 for Abstracts), <xref ref-type="supplementary-material" rid="app3">Checklist 2</xref> (PRISMA 2020 expanded checklist), and <xref ref-type="supplementary-material" rid="app4">Checklist 3</xref> (PRISMA-S extension), with each item mapped to the corresponding manuscript location. The protocol was prospectively registered with the International Prospective Register of Systematic Reviews (PROSPERO; CRD420261284298, registered on January 13, 2026). Patients or the public were not involved in the design, conduct, reporting, or dissemination plans of our research.</p></sec><sec id="s2-2"><title>Eligibility Criteria</title><p>Eligibility was determined using the population, intervention, comparator, and outcomes framework [<xref ref-type="bibr" rid="ref21">21</xref>]. Eligible populations included health care professionals actively engaged in clinical practice or clinical research, whereas medical students and nonclinical administrative staff were excluded. Eligible interventions consisted of AI-powered clinical tools, such as ambient AI documentation systems, CDSS, diagnostic AI, large language models (LLMs), predictive AI, triage systems, and AI-based burnout intervention applications. The comparator included traditional workflows without AI assistance, preimplementation periods, or no comparator for single-arm studies.</p><p>Primary outcomes focused on cognitive workload, mental fatigue, and burnout, provided they were measured with validated instruments. Acceptable cognitive workload measures included the NASA-TLX [<xref ref-type="bibr" rid="ref22">22</xref>], Subjective Workload Assessment Technique [<xref ref-type="bibr" rid="ref23">23</xref>], or Paas Cognitive Load Scale [<xref ref-type="bibr" rid="ref24">24</xref>]. Burnout had to be evaluated using established tools such as the Maslach Burnout Inventory (MBI) [<xref ref-type="bibr" rid="ref25">25</xref>], Oldenburg Burnout Inventory (OLBI) [<xref ref-type="bibr" rid="ref26">26</xref>], Stanford PFI [<xref ref-type="bibr" rid="ref27">27</xref>], Copenhagen Burnout Inventory (CBI) [<xref ref-type="bibr" rid="ref28">28</xref>], or Mini-Z [<xref ref-type="bibr" rid="ref29">29</xref>]. Secondary outcomes included alert fatigue, automation bias, trust calibration, verification burden, and System Usability Scale [<xref ref-type="bibr" rid="ref30">30</xref>].</p><p>Eligible study designs included randomized controlled trials (RCTs), non-RCTs, prospective and retrospective cohort studies, cross-sectional studies, and pre- and postimplementation studies. While all eligible studies entered the qualitative narrative synthesis, meta-analytic pooling was deliberately restricted to within-AI-category subsets to avoid combining functionally heterogeneous tools. Consequently, all 6 prespecified meta-analytic pools were drawn from the ambient AI documentation subset. Studies evaluating diagnostic imaging AI, CDSS, LLM inbox tools, and AI-based burnout interventions were reported in the narrative synthesis only and were not pooled together. Qualitative-only studies, editorials, commentaries, conference abstracts without full text, and systematic reviews were excluded. Studies measuring only time-based efficiency outcomes without validated workload instruments, studies assessing only diagnostic accuracy, and those evaluating nonclinical AI applications were also excluded.</p></sec><sec id="s2-3"><title>Information Sources</title><p>Our systematic search covered 4 electronic databases spanning January 2015 to January 2026: MEDLINE (via PubMed), Embase (via OVID), Cochrane CENTRAL, and Web of Science Core Collection. Additional sources included manual searching of reference lists, forward citation tracking, and screening of preprint servers (medRxiv and arXiv) for recent studies not yet indexed in bibliographic databases.</p></sec><sec id="s2-4"><title>Search Strategy</title><p>The search strategy combined 3 concepts using Boolean operators: AI and clinical AI tools, health care professionals, and cognitive workload or burnout. The complete database-specific search strategies are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-5"><title>Selection Process</title><p>Search results were imported into EndNote (version 21; Clarivate Analytics) and deduplicated. Two reviewers (CSB and EJG) independently screened titles and abstracts against eligibility criteria, followed by an independent full-text assessment. Disagreements were resolved through discussion or by consulting a third reviewer (JJL) if consensus could not be reached. The selection process was documented using a PRISMA 2020 flow diagram.</p></sec><sec id="s2-6"><title>Data Collection Process and Data Items</title><p>Data were extracted independently by 2 reviewers using a standardized extraction form. For studies with multiple intervention groups or time points, data from all relevant arms were extracted. Authors were contacted for clarification or additional data when necessary. The extracted information included study characteristics (authors, year, country, journal, and study design), population characteristics (sample size, participant type, clinical specialty, and experience level), intervention details (AI system type, clinical domain, and implementation setting), outcome measures (validated instruments used, outcome definitions, and measurement time points), and results (effect estimates, CIs, <italic>P</italic> values, and pre-post comparisons).</p></sec><sec id="s2-7"><title>Study Risk of Bias Assessment</title><p>Risk of bias was assessed independently by 2 reviewers using domain-appropriate tools. For RCTs, we used the Cochrane Risk of Bias tool version 2.0 (RoB 2.0) [<xref ref-type="bibr" rid="ref31">31</xref>], which evaluates bias arising from the randomization process, deviations from intended interventions, missing outcome data, measurement of the outcome, and selection of reported results. For non-RCTs, we used the Risk of Bias in Non-randomized Studies of Interventions (ROBINS-I) tool [<xref ref-type="bibr" rid="ref32">32</xref>], which evaluates bias due to confounding, selection of participants, classification of interventions, deviations from intended interventions, missing data, measurement of outcomes, and selection of reported results. Each domain was rated as &#x201C;low risk,&#x201D; &#x201C;some concerns,&#x201D; or &#x201C;high risk&#x201D; for RoB 2.0, and &#x201C;low,&#x201D; &#x201C;moderate,&#x201D; &#x201C;serious,&#x201D; or &#x201C;critical&#x201D; risk for ROBINS-I. Disagreements were resolved by discussion.</p></sec><sec id="s2-8"><title>Effect Measures</title><p>We calculated standardized mean differences (SMDs and Hedges <italic>g</italic>) for continuous outcomes measured with different scales (eg, NASA-TLX subscales using 0&#x2010;10 vs 0&#x2010;20 ranges). For continuous outcomes measured with the same instrument, mean differences (MDs) were used. Odds ratios (ORs) were calculated for binary outcomes (eg, burnout prevalence). For pre-post studies, a within-individual correlation of <italic>r</italic>=0.7 was assumed between baseline and follow-up, accompanied by sensitivity analyses were performed at <italic>r</italic>=0.5 and <italic>r</italic>=0.9.</p></sec><sec id="s2-9"><title>Synthesis Methods</title><p>We adopted a narrative synthesis as the primary integrative approach due to anticipated heterogeneity across AI applications, clinical domains, outcome measures, and study designs [<xref ref-type="bibr" rid="ref33">33</xref>]. Where 2 or more studies reported the same outcome using the same validated instrument in comparable contexts, random-effects meta-analysis was performed. No participant overlap exists across the 6 prespecified meta-analytic pools. NASA-TLX data were pooled on a subscale-by-subscale basis (mental demand, temporal demand, and effort) using Hedges <italic>g</italic> to standardize different scale ranges (0&#x2010;10 and 0&#x2010;20). We explicitly avoided pooling aggregate scores from modified versions of the instrument, combining only studies that reported identical subscales on comparable measurement structures. This pooling strategy introduces measurement heterogeneity, which is addressed in the Limitations section. Statistical heterogeneity was assessed using <italic>I</italic><sup>2</sup> (&#x003C;25%, 25%&#x2010;75%, &#x003E;75% for low, moderate, and high, respectively), complemented by &#x03C4;<sup>2</sup>. Subgroup analyses were planned by study design, health care professional type, AI system type, and geographic region; however, the small number of studies per outcome precluded these analyses.</p></sec><sec id="s2-10"><title>Statistical Synthesis Methods</title><p>Random-effects meta-analyses applied the Hartung-Knapp-Sidik-Jonkman (HKSJ) adjustment [<xref ref-type="bibr" rid="ref34">34</xref>] uniformly as the primary inferential method for all 6 prespecified pools, irrespective of the number of contributing studies. The between-study variance &#x03C4;<sup>2</sup> was estimated by restricted maximum likelihood. We constructed Knapp-Hartung CIs using the 2-tailed <italic>t</italic> distribution. The variance-inflation factor <italic>q</italic>* was truncated to a minimum of 1 to ensure the adjusted interval remained at least as wide as the unadjusted random-effects CIs [<xref ref-type="bibr" rid="ref34">34</xref>]. Prediction intervals (PIs) were calculated for pools containing 3 or more studies [<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref36">36</xref>]. For pools with only 2 studies, we reported the pooled point estimate with its 95% Knapp-Hartung CI without a PI. This conservative approach intentionally trades narrow-band precision for protection against type I error inflation.</p><p>CIs are used to quantify the precision of the average pooled effect, whereas PIs estimate the expected distribution of true effects across new settings. These metrics are reported alongside each other where estimable. We confirmed that no participants were shared between studies in any meta-analytic pool. The NASA-TLX subscale pools [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref10">10</xref>], the PFI work exhaustion pool [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref37">37</xref>], the burnout prevalence pool [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref38">38</xref>], and the documentation time pool [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref39">39</xref>] all combine nonoverlapping samples; therefore, no correlation adjustment for shared participants was required. Where pre-post designs contributed continuous outcomes, we calculated effect sizes using change-score SDs derived from pooled pre and post SDs assuming a pre-post correlation of <italic>r</italic>=0.7 with sensitivity analyses at <italic>r</italic>=0.5 and <italic>r</italic>=0.9. For burnout prevalence, included studies [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref38">38</xref>] reported paired pre-post counts within the same cohort, and paired ORs were computed (or derived from pooled pre or post 2&#x00D7;2 tables where individual-level data were unavailable). This paired approach is more conservative than treating proportions as independent because it accounts for within-individual correlation.</p></sec><sec id="s2-11"><title>Reporting Bias Assessment</title><p>We did not perform formal assessments of reporting bias using funnel plots or Egger test because all pools contained fewer than the recommended threshold of 10 studies [<xref ref-type="bibr" rid="ref40">40</xref>]. Instead, the potential for small-study effects was considered qualitatively within the GRADE (Grading of Recommendations Assessment, Development and Evaluation) assessment.</p></sec><sec id="s2-12"><title>Certainty Assessment</title><p>The certainty of evidence for each outcome was assessed using the GRADE approach [<xref ref-type="bibr" rid="ref41">41</xref>]. Evidence was rated as high, moderate, low, or very low based on 5 domains: risk of bias, inconsistency, indirectness, imprecision, and publication bias. GRADE-Confidence in the Evidence from Reviews of Qualitative Research for qualitative findings was prespecified in the protocol but could not be performed because no qualitative-only study was identified.</p></sec><sec id="s2-13"><title>Reporting Standards and Protocol Deviations</title><p>Reporting standards and the corresponding completed checklists are described in <xref ref-type="supplementary-material" rid="app2">Checklists 1</xref><xref ref-type="supplementary-material" rid="app3"/>-<xref ref-type="supplementary-material" rid="app4">3</xref>. The protocol deviations described in this section reflect refinements made during the conduct of the review and are reported transparently in keeping with PRISMA 2020 item 5. The systematic review was prospectively registered in PROSPERO prior to data extraction (CRD420261284298, registered on January 13, 2026).</p><p>We acknowledge the following protocol deviations. First, a lower-bound search date restriction of January 2015 was applied to focus on the contemporary clinical AI era. Second, medical students were excluded because their clinical exposure to AI tools differs substantially from that of practicing clinicians. Third, qualitative-only studies were excluded, which consequently omitted the planned thematic synthesis and GRADE-Confidence in the Evidence from Reviews of Qualitative Research assessments. Fourth, we additionally searched Cochrane CENTRAL to capture trials potentially unindexed elsewhere. Fifth, burnout was elevated to a coprimary outcome due to its prevalence in the identified literature. Sixth, several prespecified secondary outcomes were omitted because no included studies measured them with validated instruments. Seventh, AI-based burnout intervention applications were retained, as they directly addressed the research question. Finally, the HKSJ adjustment was uniformly adopted as the primary inferential method for all meta-analytic pools, replacing the DerSimonian-Laird estimator.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Study Selection</title><p>The systematic search identified 8848 records across 4 databases: PubMed or MEDLINE (n=2055), Embase or Ovid (n=6024), Cochrane CENTRAL (n=65), and Web of Science (n=704). After deduplication in EndNote, 1520 duplicate records were removed. This left 7328 unique records that underwent title or abstract screening, of which 7019 records were excluded. The remaining 309 reports were retrieved and assessed for full-text eligibility. We excluded 288 papers for specific reasons, including narrative review (n=10), study with incomplete data (n=277), and systematic review (n=1). Furthermore, we excluded 2 large ambient AI scribe studies because they did not meet the validated instrument criterion. Specifically, Tierney et al [<xref ref-type="bibr" rid="ref5">5</xref>] (n=7260) used custom satisfaction surveys without NASA-TLX or standardized burnout measures, and Albrecht et al [<xref ref-type="bibr" rid="ref6">6</xref>] (n=181) relied on a custom quality improvement survey without validated workload assessments. In total, 21 studies met all inclusion criteria and were included in the qualitative synthesis. The study selection process is presented in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) 2020 flow diagram of the study selection process.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e93618_fig01.png"/></fig></sec><sec id="s3-2"><title>Study Characteristics</title><p>The 21 included studies [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref37">37</xref>-<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref42">42</xref>-<xref ref-type="bibr" rid="ref52">52</xref>] were published between 2019 and 2025. In total, 19 (90.5%) of these studies [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref39">39</xref>-<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref48">48</xref>-<xref ref-type="bibr" rid="ref52">52</xref>] were published in 2024&#x2010;2025, indicating recent expansion in this research area. Studies were conducted across 7 countries: the United States (15/21, 71.4%) [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref37">37</xref>-<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref42">42</xref>-<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref49">49</xref>], Germany (1/21, 4.8%) [<xref ref-type="bibr" rid="ref16">16</xref>], Australia (1/21, 4.8%) [<xref ref-type="bibr" rid="ref50">50</xref>], South Korea (1/21, 4.8%) [<xref ref-type="bibr" rid="ref51">51</xref>], the United Kingdom (1/21, 4.8%) [<xref ref-type="bibr" rid="ref46">46</xref>], Zambia (1/21, 4.8%) [<xref ref-type="bibr" rid="ref48">48</xref>], and the United Arab Emirates (1/21, 4.8%) [<xref ref-type="bibr" rid="ref52">52</xref>]. The total number of health care professionals across all studies was 2885, with sample sizes ranging from 7 to 1430 participants.</p><p>Regarding study design, 3 studies were RCTs [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref51">51</xref>], which consisted of 1 parallel 3-arm RCT [<xref ref-type="bibr" rid="ref7">7</xref>], 1 stepped-wedge RCT [<xref ref-type="bibr" rid="ref8">8</xref>], and 1 single-blind 3-group RCT [<xref ref-type="bibr" rid="ref51">51</xref>]. In total, 3 further studies used randomized crossover designs [<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref52">52</xref>], 2 were randomized simulation studies [<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref49">49</xref>], and 13 were observational studies [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref37">37</xref>-<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref48">48</xref>]. The observational group included pre-post designs (n=9) [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref37">37</xref>-<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref45">45</xref>], cross-sectional studies (n=2) [<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref47">47</xref>], quality improvement studies (n=1) [<xref ref-type="bibr" rid="ref4">4</xref>], and mixed methods study (n=1) [<xref ref-type="bibr" rid="ref48">48</xref>].</p><p>Participants included physicians (n=19) [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref37">37</xref>-<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref42">42</xref>-<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref52">52</xref>], advanced practice providers (APPs; n=9) [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref8">8</xref>-<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref43">43</xref>-<xref ref-type="bibr" rid="ref45">45</xref>], nurses (n=3) [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref51">51</xref>], and radiology trainees (n=2) [<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref48">48</xref>]. The evaluated clinical domains covered outpatient documentation (n=12) [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref37">37</xref>-<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref42">42</xref>-<xref ref-type="bibr" rid="ref45">45</xref>], diagnostic imaging (n=3) [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref48">48</xref>], pediatric clinical care (n=4) [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref49">49</xref>], trauma and transfusion (n=1) [<xref ref-type="bibr" rid="ref50">50</xref>], inpatient documentation (n=1) [<xref ref-type="bibr" rid="ref46">46</xref>], nursing practice (n=1) [<xref ref-type="bibr" rid="ref51">51</xref>], and psychiatric practice (n=1) [<xref ref-type="bibr" rid="ref52">52</xref>].</p><p>AI interventions were categorized into 5 types: ambient AI documentation systems (n=13) [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref42">42</xref>-<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref52">52</xref>], CDSS (n=3) [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref50">50</xref>], radiology AI (n=3) [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref48">48</xref>], LLM-based inbox management (n=1) [<xref ref-type="bibr" rid="ref37">37</xref>], and AI-based burnout intervention (n=1) [<xref ref-type="bibr" rid="ref51">51</xref>]. The characteristics of included studies are summarized in <xref ref-type="table" rid="table1">Table 1</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Characteristics of the 21 included studies.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study</td><td align="left" valign="bottom">Country</td><td align="left" valign="bottom">Study design</td><td align="left" valign="bottom">Setting</td><td align="left" valign="bottom">Values, n</td><td align="left" valign="bottom">Participants</td><td align="left" valign="bottom">AI<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> system</td><td align="left" valign="bottom">Instruments</td><td align="left" valign="bottom">Follow-up duration</td><td align="left" valign="bottom">Primary outcome</td><td align="left" valign="bottom">Key findings</td><td align="left" valign="bottom">Risk of bias</td><td align="left" valign="bottom">Interpretation</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="13">Ambient AI documentation (n=13)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Lukac et al (2025) [<xref ref-type="bibr" rid="ref7">7</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Parallel 3-arm pragmatic RCT<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">Multispecialty clinic (outpatient)</td><td align="left" valign="top">238</td><td align="left" valign="top">Physicians</td><td align="left" valign="top">DAX<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup> Copilot; Nabla</td><td align="left" valign="top">Mini-Z 2.0, 4-item PTL<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup>, PFI<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup>-WE<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup></td><td align="left" valign="top">12 weeks</td><td align="left" valign="top">Task load, WE</td><td align="left" valign="top">PTL: DAX &#x2212;39.9, Nabla &#x2212;31.7 (<italic>P</italic>&#x003C;.01); PFI-WE &#x2212;0.27 (0&#x2010;4 scale; <italic>P</italic>=.01); documentation time: Nabla &#x2212;9.5%</td><td align="left" valign="top">Low</td><td align="left" valign="top">Ambient AI significantly reduces physician task load across multiple specialties</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Afshar et al (2025) [<xref ref-type="bibr" rid="ref8">8</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Stepped-wedge RCT</td><td align="left" valign="top">Primary care (outpatient)</td><td align="left" valign="top">66</td><td align="left" valign="top">Physicians, APPs<sup><xref ref-type="table-fn" rid="table1fn7">g</xref></sup></td><td align="left" valign="top">Abridge</td><td align="left" valign="top">Stanford PFI, PDQI-9<sup><xref ref-type="table-fn" rid="table1fn8">h</xref></sup></td><td align="left" valign="top">24 weeks</td><td align="left" valign="top">WE</td><td align="left" valign="top">WE/ID<sup><xref ref-type="table-fn" rid="table1fn9">i</xref></sup>: &#x2212;0.44 (95% CI &#x2212;0.62 to &#x2212;0.25; <italic>P</italic>&#x003C;.001)</td><td align="left" valign="top">Some concerns</td><td align="left" valign="top">Reduction in burnout with sustained 6-month follow-up</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Olson et al (2025) [<xref ref-type="bibr" rid="ref4">4</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Multicenter quality improvement pre-post</td><td align="left" valign="top">Multisite (outpatient)</td><td align="left" valign="top">263</td><td align="left" valign="top">Physicians, APPs</td><td align="left" valign="top">Abridge</td><td align="left" valign="top">Single-item burnout, 3-item NASA-TLX<sup><xref ref-type="table-fn" rid="table1fn10">j</xref></sup></td><td align="left" valign="top">3&#x2010;6 months</td><td align="left" valign="top">Burnout prevalence</td><td align="left" valign="top">Burnout: 51.9%&#x2192;38.8% (aOR<sup><xref ref-type="table-fn" rid="table1fn11">k</xref></sup> 0.26; <italic>P</italic>&#x003C;.001); TLX<sup><xref ref-type="table-fn" rid="table1fn12">l</xref></sup>: &#x2212;2.64</td><td align="left" valign="top">Serious</td><td align="left" valign="top">13 pp<sup><xref ref-type="table-fn" rid="table1fn14">n</xref></sup> absolute burnout reduction, single-arm pre- and postdesign precludes causal or scalability claims</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Shah et al (2025) [<xref ref-type="bibr" rid="ref42">42</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Prospective quality improvement pre-post</td><td align="left" valign="top">Academic medical center (outpatient)</td><td align="left" valign="top">48</td><td align="left" valign="top">Physicians</td><td align="left" valign="top">DAX Copilot</td><td align="left" valign="top">NASA-TLX, SUS<sup><xref ref-type="table-fn" rid="table1fn13">m</xref></sup>, PFI</td><td align="left" valign="top">8 weeks</td><td align="left" valign="top">Cognitive workload</td><td align="left" valign="top">NASA-TLX: &#x2212;24.42 (<italic>P</italic>&#x003C;.001); burnout: &#x2212;1.94; SUS: +10.9</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Large NASA-TLX reduction (24 points) observed, with high usability; uncontrolled pre- and postdesign</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hudson et al (2025) [<xref ref-type="bibr" rid="ref43">43</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Crossover RCT</td><td align="left" valign="top">Academic medical center (outpatient)</td><td align="left" valign="top">40</td><td align="left" valign="top">Physicians, APPs</td><td align="left" valign="top">Abridge</td><td align="left" valign="top">NASA-TLX</td><td align="left" valign="top">2 weeks</td><td align="left" valign="top">Cognitive load</td><td align="left" valign="top">NASA-TLX: 221&#x2192;118 (&#x2212;60.7%; <italic>P</italic>&#x003C;.001); mental &#x2212;57%</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">60% reduction represents largest effect observed</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>You et al (2025) [<xref ref-type="bibr" rid="ref9">9</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Multisite pre-post</td><td align="left" valign="top">Academic medical center (outpatient)</td><td align="left" valign="top">1430</td><td align="left" valign="top">Physicians, APPs</td><td align="left" valign="top">Multiple ambient AI systems</td><td align="left" valign="top">Stanford PFI</td><td align="left" valign="top">42 days</td><td align="left" valign="top">Burnout, well-being</td><td align="left" valign="top">Burnout: &#x2212;21.2 pp<sup><xref ref-type="table-fn" rid="table1fn14">n</xref></sup> (<italic>P</italic>&#x003C;.001)</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Largest sample size; 21 pp reduction at 42 days in the MGB<sup><xref ref-type="table-fn" rid="table1fn15">o</xref></sup> cohort; no control group</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Owens et al (2024) [<xref ref-type="bibr" rid="ref44">44</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Cross-sectional</td><td align="left" valign="top">Primary care</td><td align="left" valign="top">110</td><td align="left" valign="top">PCPs<sup><xref ref-type="table-fn" rid="table1fn16">p</xref></sup></td><td align="left" valign="top">DAX</td><td align="left" valign="top">OLBI<sup><xref ref-type="table-fn" rid="table1fn17">q</xref></sup></td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table1fn18">r</xref></sup></td><td align="left" valign="top">Burnout</td><td align="left" valign="top">Disengagement: &#x2212;2.1 (<italic>P</italic>&#x003C;.05)</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Higher AI use associated with lower disengagement in cross-sectional comparison; causal direction cannot be inferred</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Misurac et al (2025) [<xref ref-type="bibr" rid="ref45">45</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Pre-post</td><td align="left" valign="top">Academic medical center (outpatient)</td><td align="left" valign="top">35</td><td align="left" valign="top">Physicians, APPs</td><td align="left" valign="top">Ambient AI (not specified)</td><td align="left" valign="top">Stanford PFI</td><td align="left" valign="top">3 months</td><td align="left" valign="top">Burnout</td><td align="left" valign="top">Burnout: 69%&#x2192;43% (<italic>P</italic>=.005)</td><td align="left" valign="top">Serious</td><td align="left" valign="top">26 pp absolute burnout reduction in an uncontrolled pre- and postsample (n=35)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Stults et al (2025) [<xref ref-type="bibr" rid="ref10">10</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Pre-post</td><td align="left" valign="top">Integrated health system (outpatient)</td><td align="left" valign="top">100</td><td align="left" valign="top">Ambulatory clinicians</td><td align="left" valign="top">Abridge</td><td align="left" valign="top">NASA-TLX, single-item burnout</td><td align="left" valign="top">12 weeks</td><td align="left" valign="top">Documentation time, workload</td><td align="left" valign="top">Time: 6.2&#x2192;5.3 minutes (<italic>P</italic>&#x003C;.001); NASA-TLX mental demand &#x2193;; burnout: 42.1%&#x2192;35.1%</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Multisite replication of ambient AI benefits; 7 pp burnout reduction consistent with other studies</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Duggan et al (2025) [<xref ref-type="bibr" rid="ref39">39</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Prospective quality improvement pre-post</td><td align="left" valign="top">Academic medical center (outpatient)</td><td align="left" valign="top">46</td><td align="left" valign="top">Physicians, APPs</td><td align="left" valign="top">DAX Copilot</td><td align="left" valign="top">NASA-TLX items, SUS</td><td align="left" valign="top">8 weeks</td><td align="left" valign="top">Documentation burden, efficiency</td><td align="left" valign="top">SUS positive; documentation burden perceived &#x2193;; mixed individual experiences</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Highlights variability in clinician experience; not all users benefit equally from AI scribes</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Pelletier et al (2025) [<xref ref-type="bibr" rid="ref38">38</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Pre-post with ITSA<sup><xref ref-type="table-fn" rid="table1fn19">s</xref></sup></td><td align="left" valign="top">Pediatric hospital (outpatient)</td><td align="left" valign="top">84</td><td align="left" valign="top">Pediatric physicians, APPs</td><td align="left" valign="top">Abridge</td><td align="left" valign="top">NASA-TLX, Mini-Z</td><td align="left" valign="top">6 months</td><td align="left" valign="top">Documentation time, workload, burnout</td><td align="left" valign="top">Time: &#x2212;2.8 minutes per appointment (<italic>P</italic>&#x003C;.001); 1.5 hours per week saved; NASA-TLX &#x2193;; Mini-Z burnout &#x2193;</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">First pediatric ambient AI study; ITSA strengthens causal inference; pediatric settings</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Bracken et al (2026) [<xref ref-type="bibr" rid="ref46">46</xref>]</td><td align="left" valign="top">United Kingdom</td><td align="left" valign="top">Simulation</td><td align="left" valign="top">Simulated inpatient ward (orthopedic surgery)</td><td align="left" valign="top">7</td><td align="left" valign="top">PGY<sup><xref ref-type="table-fn" rid="table1fn20">t</xref></sup>-1 doctors</td><td align="left" valign="top">Heidi Health</td><td align="left" valign="top">NASA-TLX, PDQI-9</td><td align="left" valign="top">1 session</td><td align="left" valign="top">Documentation time</td><td align="left" valign="top">Time: 27 versus 128 seconds; frustration &#x2212;79%</td><td align="left" valign="top">Critical</td><td align="left" valign="top">Large simulated time savings; n=7, simulation only&#x2014;critical risk of bias</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Nawaz et al (2025) [<xref ref-type="bibr" rid="ref52">52</xref>]</td><td align="left" valign="top">United Arab Emirates</td><td align="left" valign="top">Crossover simulation</td><td align="left" valign="top">Psychiatric hospital (simulation)</td><td align="left" valign="top">8</td><td align="left" valign="top">Psychiatrists</td><td align="left" valign="top">Lyrebird Health</td><td align="left" valign="top">NASA-TLX, SAIL<sup><xref ref-type="table-fn" rid="table1fn21">u</xref></sup></td><td align="left" valign="top">1 session</td><td align="left" valign="top">Workload, documentation quality</td><td align="left" valign="top">NASA-TLX: 25.0 versus 461.3 (<italic>P</italic>&#x003C;.001); largest effect in review; SAIL quality &#x2191;</td><td align="left" valign="top">Critical</td><td align="left" valign="top">Preprint; largest effect in this review but n=8, simulation only&#x2014;critical risk of bias; first psychiatric ambient AI study</td></tr><tr><td align="left" valign="top" colspan="13">Diagnostic imaging AI (n=3<bold>)</bold></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Wenderott et al (2024) [<xref ref-type="bibr" rid="ref16">16</xref>]</td><td align="left" valign="top">Germany</td><td align="left" valign="top">Prospective pre-post</td><td align="left" valign="top">Academic medical center</td><td align="left" valign="top">91 cases</td><td align="left" valign="top">Radiologists</td><td align="left" valign="top">Prostate MRI<sup><xref ref-type="table-fn" rid="table1fn22">v</xref></sup> CADe<sup><xref ref-type="table-fn" rid="table1fn23">w</xref></sup> (commercial, not specified)</td><td align="left" valign="top">NASA-TLX, STAI<sup><xref ref-type="table-fn" rid="table1fn24">x</xref></sup></td><td align="left" valign="top">6 months</td><td align="left" valign="top">Workload, reading time</td><td align="left" valign="top">NASA-TLX: NS<sup><xref ref-type="table-fn" rid="table1fn25">y</xref></sup> (<italic>P</italic>=.51); time &#x2191;15.7&#x2192;23.1 minutes (<italic>P</italic>=.02)</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Critical negative finding: diagnostic AI may increase rather than decrease cognitive demands</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Lopez-Rippe et al (2025) [<xref ref-type="bibr" rid="ref47">47</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Mixed methods</td><td align="left" valign="top">Academic medical center</td><td align="left" valign="top">NR<sup><xref ref-type="table-fn" rid="table1fn26">z</xref></sup></td><td align="left" valign="top">Radiology trainees</td><td align="left" valign="top">RADHawk</td><td align="left" valign="top">NASA-TLX</td><td align="left" valign="top">6 months</td><td align="left" valign="top">Cognitive load</td><td align="left" valign="top">Reduced workload and mental demand (values NR)</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Knowledge-based AI support may reduce trainee burden, but quantitative data lacking</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Muzumala et al (2025) [<xref ref-type="bibr" rid="ref48">48</xref>]</td><td align="left" valign="top">Zambia</td><td align="left" valign="top">Comparative experiment</td><td align="left" valign="top">Pediatric hospital</td><td align="left" valign="top">12</td><td align="left" valign="top">Radiology residents</td><td align="left" valign="top">Pneumonia chest x-ray diagnosis AI (not specified)</td><td align="left" valign="top">NASA-TLX, TAM<sup><xref ref-type="table-fn" rid="table1fn27">aa</xref></sup>-2</td><td align="left" valign="top">1 session</td><td align="left" valign="top">Workload, usability</td><td align="left" valign="top">NASA-TLX: 1.86 (7-point scale); positive TAM-2</td><td align="left" valign="top">Serious</td><td align="left" valign="top">Low-resource setting feasibility demonstrated; modified scale limits cross-study comparison</td></tr><tr><td align="left" valign="top" colspan="13">CDSS<sup><xref ref-type="table-fn" rid="table1fn28">ab</xref></sup> (n=3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Richardson et al (2019) [<xref ref-type="bibr" rid="ref49">49</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Crossover simulation</td><td align="left" valign="top">Pediatric hospital</td><td align="left" valign="top">32</td><td align="left" valign="top">Pediatric physicians</td><td align="left" valign="top">PedsGuide (Children&#x2019;s Mercy)</td><td align="left" valign="top">NASA-TLX, SUS</td><td align="left" valign="top">1 session</td><td align="left" valign="top">Mental workload</td><td align="left" valign="top">Mental: 6.34 versus 11.8 (<italic>P</italic>&#x003C;.001); SUS: 88/100</td><td align="left" valign="top">Low</td><td align="left" valign="top">Well-designed CDSS can halve mental demand while improving clinical performance</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Kandaswamy et al (2025) [<xref ref-type="bibr" rid="ref17">17</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Mixed methods</td><td align="left" valign="top">Pediatric emergency department</td><td align="left" valign="top">40</td><td align="left" valign="top">Clinicians, nurses</td><td align="left" valign="top">IPSO<sup><xref ref-type="table-fn" rid="table1fn29">ac</xref></sup> sepsis AI (CHOA<sup><xref ref-type="table-fn" rid="table1fn30">ad</xref></sup>)</td><td align="left" valign="top">NASA-TLX, SUS</td><td align="left" valign="top">6 months</td><td align="left" valign="top">Workload, trust</td><td align="left" valign="top">NASA-TLX: 43&#x2192;57 (workload &#x2191;); trust: 3.8/5</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Alert-based AI paradoxically increases workload; highlights verification burden and alert fatigue</td></tr><tr><td align="left" valign="top">&#x2003;Sanderson et al (2023) [<xref ref-type="bibr" rid="ref50">50</xref>]</td><td align="left" valign="top">Australia</td><td align="left" valign="top">Crossover simulation</td><td align="left" valign="top">Pediatric hospital</td><td align="left" valign="top">44</td><td align="left" valign="top">Physicians, nurses</td><td align="left" valign="top">MT<sup><xref ref-type="table-fn" rid="table1fn31">ae</xref></sup>-CDS<sup><xref ref-type="table-fn" rid="table1fn32">af</xref></sup> (Westmead Hospital)</td><td align="left" valign="top">NASA-TLX, SUS</td><td align="left" valign="top">1 session</td><td align="left" valign="top">Workload, decisions</td><td align="left" valign="top">NASA-TLX: CDS 57.1 versus paper 64.5 (<italic>P</italic>=.005); SUS: 82.5</td><td align="left" valign="top">Low</td><td align="left" valign="top">CDS reduces workload in high-stakes decisions while improving decision velocity</td></tr><tr><td align="left" valign="top" colspan="13">Large language model (LLM)&#x2013;based tools (n=1)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Garcia et al (2024) [<xref ref-type="bibr" rid="ref37">37</xref>]</td><td align="left" valign="top">United States</td><td align="left" valign="top">Prospective quality improvement pre-post</td><td align="left" valign="top">Hospital</td><td align="left" valign="top">162</td><td align="left" valign="top">Physicians</td><td align="left" valign="top">GPT-4 (OpenAI)</td><td align="left" valign="top">PTL, PFI-WE</td><td align="left" valign="top">5 weeks</td><td align="left" valign="top">Task load</td><td align="left" valign="top">PTL: 61.3&#x2192;47.3 (&#x2212;13.9; <italic>P</italic>&#x003C;.001); WE: &#x2212;0.33</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">Early LLM evidence shows promise for inbox management; 20% adoption rate indicates acceptability</td></tr><tr><td align="left" valign="top" colspan="13">AI burnout intervention (n=1)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Baek and Cha (2025) [<xref ref-type="bibr" rid="ref51">51</xref>]</td><td align="left" valign="top">Korea</td><td align="left" valign="top">3-group RCT</td><td align="left" valign="top">Hospital</td><td align="left" valign="top">120</td><td align="left" valign="top">Nurses</td><td align="left" valign="top">Nurse Healing Space (AI burnout application)</td><td align="left" valign="top">CBI<sup><xref ref-type="table-fn" rid="table1fn33">ag</xref></sup></td><td align="left" valign="top">4 weeks</td><td align="left" valign="top">Burnout</td><td align="left" valign="top">Client burnout: 62.6&#x2192;42.0 (<italic>P</italic>=.001); personal: 67.5&#x2192;44.7</td><td align="left" valign="top">Low</td><td align="left" valign="top">AI-tailored intervention reduced nursing burnout in an RCT; zero dropout</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>AI: artificial intelligence.</p></fn><fn id="table1fn2"><p><sup>b</sup>RCT: randomized controlled trial.</p></fn><fn id="table1fn3"><p><sup>c</sup>DAX: Dragon Ambient Experience.</p></fn><fn id="table1fn4"><p><sup>d</sup>PTL: Physician Task Load.</p></fn><fn id="table1fn5"><p><sup>e</sup>PFI: Professional Fulfillment Index.</p></fn><fn id="table1fn6"><p><sup>f</sup>WE: Work Exhaustion.</p></fn><fn id="table1fn7"><p><sup>g</sup>APP: advanced practice provider.</p></fn><fn id="table1fn8"><p><sup>h</sup>PDQI-9: Physician Documentation Quality Instrument.</p></fn><fn id="table1fn9"><p><sup>i</sup>ID: interpersonal disengagement.</p></fn><fn id="table1fn10"><p><sup>j</sup>NASA-TLX: National Aeronautics and Space Administration Task Load Index.</p></fn><fn id="table1fn11"><p><sup>k</sup>aOR: adjusted odds ratio.</p></fn><fn id="table1fn12"><p><sup>l</sup>TLX: Task Load Index.</p></fn><fn id="table1fn13"><p><sup>m</sup>SUS: System Usability Scale.</p></fn><fn id="table1fn14"><p><sup>n</sup>pp: percentage point.</p></fn><fn id="table1fn15"><p><sup>o</sup>MGB: Mass General Brigham.</p></fn><fn id="table1fn16"><p><sup>p</sup>PCP: primary care physician.</p></fn><fn id="table1fn17"><p><sup>q</sup>OLBI: Oldenburg Burnout Inventory.</p></fn><fn id="table1fn18"><p><sup>r</sup>Not available.</p></fn><fn id="table1fn19"><p><sup>s</sup>ITSA: interrupted Time Series Analysis.</p></fn><fn id="table1fn20"><p><sup>t</sup>PGY: postgraduate year.</p></fn><fn id="table1fn21"><p><sup>u</sup>SAIL: Sheffield Assessment Instrument for Letters.</p></fn><fn id="table1fn22"><p><sup>v</sup>MRI: magnetic resonance imaging.</p></fn><fn id="table1fn23"><p><sup>w</sup>CADe: computer-aided detection.</p></fn><fn id="table1fn24"><p><sup>x</sup>STAI: State-Trait Anxiety Inventory.</p></fn><fn id="table1fn25"><p><sup>y</sup>NS: not significant.</p></fn><fn id="table1fn26"><p><sup>z</sup>NR: not reported.</p></fn><fn id="table1fn27"><p><sup>aa</sup>TAM: Technology Acceptance Model.</p></fn><fn id="table1fn28"><p><sup>ab</sup>CDSS: clinical decision support system.</p></fn><fn id="table1fn29"><p><sup>ac</sup>IPSO: Improving Pediatric Sepsis Outcomes.</p></fn><fn id="table1fn30"><p><sup>ad</sup>CHOA: Children&#x2019;s Healthcare of Atlanta.</p></fn><fn id="table1fn31"><p><sup>ae</sup>MT: massive transfusion.</p></fn><fn id="table1fn32"><p><sup>af</sup>CDS: clinical decision support.</p></fn><fn id="table1fn33"><p><sup>ag</sup>CBI: Copenhagen Burnout Inventory.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Validated Instruments Used</title><p>The NASA-TLX or its derivative instruments were used in 16 (76%) studies [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref37">37</xref>-<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref46">46</xref>-<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref52">52</xref>]. These applications included the full 6-subscale version [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref52">52</xref>] or abbreviated versions, such as the 4-item Physician Task Load [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref37">37</xref>] and the 3-item NASA-TLX [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref43">43</xref>]. Pelletier et al [<xref ref-type="bibr" rid="ref38">38</xref>] used the NASA-TLX without specifying subscale details. Burnout-specific instruments included the Stanford PFI (n=5) [<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref45">45</xref>], Mini-Z (n=3) [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref38">38</xref>], OLBI (n=1) [<xref ref-type="bibr" rid="ref44">44</xref>], CBI (n=1) [<xref ref-type="bibr" rid="ref51">51</xref>], and single-item burnout measures (n=2) [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. No study used the full MBI despite its status as the most widely validated burnout measure [<xref ref-type="bibr" rid="ref25">25</xref>]. The System Usability Scale was used as a secondary outcome in 5 studies [<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref50">50</xref>], and the Sheffield Assessment Instrument for Letters was used in 1 study to assess documentation quality [<xref ref-type="bibr" rid="ref52">52</xref>].</p></sec><sec id="s3-4"><title>Risk of Bias Assessment</title><p>Risk of bias varied substantially across studies. Among the RCTs, the 3-arm RCT by Lukac et al [<xref ref-type="bibr" rid="ref7">7</xref>] was rated as having &#x201C;low risk&#x201D; across all domains. The stepped-wedge RCT by Afshar et al [<xref ref-type="bibr" rid="ref8">8</xref>] was rated as having &#x201C;some concerns&#x201D; due to potential period effects and lack of blinding inherent to the intervention. The 3-group RCT by Baek and Cha [<xref ref-type="bibr" rid="ref51">51</xref>] was rated as &#x201C;low risk&#x201D; with zero dropouts and appropriate randomization.</p><p>Among non-RCTs assessed using ROBINS-I [<xref ref-type="bibr" rid="ref32">32</xref>], 2 studies were rated as &#x201C;low&#x201D; risk of bias [<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref50">50</xref>], 6 were rated as &#x201C;moderate&#x201D; risk of bias [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref47">47</xref>], 8 were rated as &#x201C;serious&#x201D; risk primarily due to confounding and selection bias [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref48">48</xref>], and 2 simulation studies were rated as &#x201C;critical&#x201D; risk due to small sample sizes (n=7 and n=8, respectively) [<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref52">52</xref>]. Common methodological limitations included lack of control groups, short follow-up periods, voluntary participation introducing selection bias, and use of abbreviated or modified versions of validated instruments without separate validation. The risk of bias assessment is presented in <xref ref-type="fig" rid="figure2">Figure 2</xref> and <xref ref-type="table" rid="table2">Tables 2</xref> and <xref ref-type="table" rid="table3">3</xref>.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Risk-of-bias summary for the 3 randomized controlled trials assessed with the Cochrane Risk of Bias tool version 2.0 (RoB 2) [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref51">51</xref>]. Green plus signs denote low risk of bias, and yellow question marks denote some concerns across the 5 RoB 2.0 domains and the overall judgment. Detailed domain-level judgments with supporting rationale are reported in <xref ref-type="table" rid="table2">Table 2</xref>.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e93618_fig02.png"/></fig><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Risk of bias assessment using the Cochrane Risk of Bias tool version 2.0 for the 3 included randomized controlled trials.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study</td><td align="left" valign="bottom">D1: randomization</td><td align="left" valign="bottom">D2: deviations from interventions</td><td align="left" valign="bottom">D3: missing outcome data</td><td align="left" valign="bottom">D4: measurement of outcome</td><td align="left" valign="bottom">D5: selection of reported result</td><td align="left" valign="bottom">Overall</td><td align="left" valign="bottom">Direction</td></tr></thead><tbody><tr><td align="left" valign="top">Lukac et al (2025) [<xref ref-type="bibr" rid="ref7">7</xref>]</td><td align="left" valign="top">Low (adequate 1:1:1 allocation, concealment)</td><td align="left" valign="top">Low (protocol followed)</td><td align="left" valign="top">Low (&#x003C;5% dropout)</td><td align="left" valign="top">Low (objective EHR<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>+ validated PROs<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup>)</td><td align="left" valign="top">Low (preregistered)</td><td align="left" valign="top">Low</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td></tr><tr><td align="left" valign="top">Afshar et al (2025) [<xref ref-type="bibr" rid="ref8">8</xref>]</td><td align="left" valign="top">Low (stepped-wedge appropriate)</td><td align="left" valign="top">Some concerns (no blinding, period effects possible)</td><td align="left" valign="top">Low (adequate retention)</td><td align="left" valign="top">Some concerns (self-reported, unblinded)</td><td align="left" valign="top">Low (all prespecified outcomes reported)</td><td align="left" valign="top">Some concerns</td><td align="left" valign="top">Favors intervention</td></tr><tr><td align="left" valign="top">Baek and Cha (2025) [<xref ref-type="bibr" rid="ref51">51</xref>]</td><td align="left" valign="top">Low (appropriate randomization)</td><td align="left" valign="top">Low (single-blind, participants blinded)</td><td align="left" valign="top">Low (0 dropouts)</td><td align="left" valign="top">Low (validated CBI<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup>)</td><td align="left" valign="top">Low (all outcomes reported)</td><td align="left" valign="top">Low</td><td align="left" valign="top">&#x2014;</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>EHR: electronic health record.</p></fn><fn id="table2fn2"><p><sup>b</sup>PRO: patient-reported outcome.</p></fn><fn id="table2fn3"><p><sup>c</sup>Not available.</p></fn><fn id="table2fn4"><p><sup>d</sup>CBI: Copenhagen Burnout Inventory.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Risk of bias assessment using ROBINS-I<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup> for the 18 nonrandomized studies.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study</td><td align="left" valign="bottom">D1: Confounding</td><td align="left" valign="bottom">D2: Selection</td><td align="left" valign="bottom">D3: Classification</td><td align="left" valign="bottom">D4: Deviations</td><td align="left" valign="bottom">D5: Missing data</td><td align="left" valign="bottom">D6: Measurement</td><td align="left" valign="bottom">D7: Selection of results</td><td align="left" valign="bottom">Overall</td></tr></thead><tbody><tr><td align="left" valign="top">Olson et al (2025) [<xref ref-type="bibr" rid="ref4">4</xref>]</td><td align="left" valign="top">Serious (no control, multisite confounders)</td><td align="left" valign="top">Moderate (voluntary early adopters)</td><td align="left" valign="top">Low (clear AI<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup> vs no AI)</td><td align="left" valign="top">Moderate (variable adoption rates across sites)</td><td align="left" valign="top">Low (complete EHR<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup> data)</td><td align="left" valign="top">Low (validated instruments)</td><td align="left" valign="top">Low (all outcomes reported)</td><td align="left" valign="top">Serious</td></tr><tr><td align="left" valign="top">Shah et al (2025) [<xref ref-type="bibr" rid="ref42">42</xref>]</td><td align="left" valign="top">Serious (no control, pre-post)</td><td align="left" valign="top">Moderate (single site, technology-savvy population)</td><td align="left" valign="top">Low (clear intervention)</td><td align="left" valign="top">Low (structured implementation)</td><td align="left" valign="top">Low (adequate completion)</td><td align="left" valign="top">Low (validated instruments)</td><td align="left" valign="top">Low (transparent reporting)</td><td align="left" valign="top">Serious</td></tr><tr><td align="left" valign="top">Hudson et al (2025) [<xref ref-type="bibr" rid="ref43">43</xref>]</td><td align="left" valign="top">Moderate (crossover controls within-subject confounding)</td><td align="left" valign="top">Moderate (single site)</td><td align="left" valign="top">Low (clear intervention)</td><td align="left" valign="top">Low (standardized protocol)</td><td align="left" valign="top">Low (complete data)</td><td align="left" valign="top">Low (validated instruments)</td><td align="left" valign="top">Low (all outcomes reported)</td><td align="left" valign="top">Moderate</td></tr><tr><td align="left" valign="top">You et al (2025) [<xref ref-type="bibr" rid="ref9">9</xref>]</td><td align="left" valign="top">Serious (no control, secular trends)</td><td align="left" valign="top">Moderate (single academic center)</td><td align="left" valign="top">Low (clear AI exposure)</td><td align="left" valign="top">Low (consistent implementation)</td><td align="left" valign="top">Moderate (62% survey response rate)</td><td align="left" valign="top">Low (validated instruments)</td><td align="left" valign="top">Low (complete reporting)</td><td align="left" valign="top">Serious</td></tr><tr><td align="left" valign="top">Owens et al (2024) [<xref ref-type="bibr" rid="ref44">44</xref>]</td><td align="left" valign="top">Serious (no control, QI<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup> design)</td><td align="left" valign="top">Serious (self-selected champions, early adopters)</td><td align="left" valign="top">Low (clear intervention)</td><td align="left" valign="top">Low (protocol followed)</td><td align="left" valign="top">Low (adequate retention)</td><td align="left" valign="top">Moderate (mixed validated or unvalidated items)</td><td align="left" valign="top">Low (all outcomes reported)</td><td align="left" valign="top">Serious</td></tr><tr><td align="left" valign="top">Misurac et al (2025) [<xref ref-type="bibr" rid="ref45">45</xref>]</td><td align="left" valign="top">Serious (no control, QI study)</td><td align="left" valign="top">Moderate (single institution)</td><td align="left" valign="top">Low (clear AI use documented)</td><td align="left" valign="top">Low (standardized rollout)</td><td align="left" valign="top">Low (complete EHR metrics)</td><td align="left" valign="top">Moderate (abbreviated NASA-TLX<sup><xref ref-type="table-fn" rid="table3fn5">e</xref></sup> without separate validation)</td><td align="left" valign="top">Low (transparent reporting)</td><td align="left" valign="top">Serious</td></tr><tr><td align="left" valign="top">Bracken et al (2026) [<xref ref-type="bibr" rid="ref46">46</xref>]</td><td align="left" valign="top">Critical (simulation not real practice, artificial conditions)</td><td align="left" valign="top">Critical (n=7 junior doctors only, selection bias)</td><td align="left" valign="top">Low (clear intervention)</td><td align="left" valign="top">Low (controlled simulation)</td><td align="left" valign="top">Low (complete data)</td><td align="left" valign="top">Low (full NASA-TLX)</td><td align="left" valign="top">Low (all outcomes reported)</td><td align="left" valign="top">Critical</td></tr><tr><td align="left" valign="top">Stults et al (2025) [<xref ref-type="bibr" rid="ref10">10</xref>]</td><td align="left" valign="top">Serious (no control, pre-post QI)</td><td align="left" valign="top">Moderate (purposively sampled champions, Sutter Health investor in Abridge)</td><td align="left" valign="top">Low (clear Abridge use)</td><td align="left" valign="top">Low (structured 12-week implementation)</td><td align="left" valign="top">Low (57% survey, 92% EHR data)</td><td align="left" valign="top">Low (self-reported NASA-TLX subscales)</td><td align="left" valign="top">Low (all outcomes reported)</td><td align="left" valign="top">Serious</td></tr><tr><td align="left" valign="top">Duggan et al (2025) [<xref ref-type="bibr" rid="ref39">39</xref>]</td><td align="left" valign="top">Serious (no control, pre-post)</td><td align="left" valign="top">Serious (voluntary enrollment, n=46 from 17 specialties, heterogeneous)</td><td align="left" valign="top">Low (clear DAX<sup><xref ref-type="table-fn" rid="table3fn6">f</xref></sup> Copilot use)</td><td align="left" valign="top">Low (8-week structured pilot)</td><td align="left" valign="top">Moderate (incomplete survey responses noted)</td><td align="left" valign="top">Moderate (self-reported SUS<sup><xref ref-type="table-fn" rid="table3fn7">g</xref></sup> and NASA-TLX items)</td><td align="left" valign="top">Low (mixed results transparently reported)</td><td align="left" valign="top">Serious</td></tr><tr><td align="left" valign="top">Pelletier et al (2025) [<xref ref-type="bibr" rid="ref38">38</xref>]</td><td align="left" valign="top">Moderate (ITSA<sup><xref ref-type="table-fn" rid="table3fn8">h</xref></sup> controls secular trends, level+ slope analysis)</td><td align="left" valign="top">Moderate (broader recruitment n=84, multiple pediatric specialties)</td><td align="left" valign="top">Low (clear Abridge use)</td><td align="left" valign="top">Low (6-month structured implementation)</td><td align="left" valign="top">Low (complete EHR metrics)</td><td align="left" valign="top">Moderate (mixed objective EHR +subjective surveys)</td><td align="left" valign="top">Low (all outcomes reported)</td><td align="left" valign="top">Moderate</td></tr><tr><td align="left" valign="top">Nawaz et al (2025) [<xref ref-type="bibr" rid="ref52">52</xref>]</td><td align="left" valign="top">Critical (simulation environment, not real clinical practice)</td><td align="left" valign="top">Critical (n=8 only, selection bias)</td><td align="left" valign="top">Low (clear Lyrebird use)</td><td align="left" valign="top">Low (controlled crossover protocol)</td><td align="left" valign="top">Low (complete data, small n)</td><td align="left" valign="top">Moderate (simulation may artificially inflate effects)</td><td align="left" valign="top">Low (all outcomes reported)</td><td align="left" valign="top">Critical</td></tr><tr><td align="left" valign="top">Wenderott et al (2024) [<xref ref-type="bibr" rid="ref16">16</xref>]</td><td align="left" valign="top">Moderate (within-subject comparison partially controls confounding)</td><td align="left" valign="top">Moderate (single radiology department)</td><td align="left" valign="top">Low (clear AI vs no AI cases)</td><td align="left" valign="top">Low (standardized reading protocol)</td><td align="left" valign="top">Low (complete case data)</td><td align="left" valign="top">Low (validated NASA-TLX)</td><td align="left" valign="top">Low (all outcomes reported)</td><td align="left" valign="top">Moderate</td></tr><tr><td align="left" valign="top">Lopez-Rippe et al (2025) [<xref ref-type="bibr" rid="ref47">47</xref>]</td><td align="left" valign="top">Moderate (crossover design reduces confounding)</td><td align="left" valign="top">Serious (convenience sample of radiology trainees)</td><td align="left" valign="top">Low (clear AI intervention)</td><td align="left" valign="top">Low (structured simulation)</td><td align="left" valign="top">Moderate (incomplete workload assessments)</td><td align="left" valign="top">Moderate (abbreviated NASA-TLX)</td><td align="left" valign="top">Low (transparent reporting)</td><td align="left" valign="top">Moderate</td></tr><tr><td align="left" valign="top">Muzumala et al (2025) [<xref ref-type="bibr" rid="ref48">48</xref>]</td><td align="left" valign="top">Serious (no control, observational only)</td><td align="left" valign="top">Moderate (single site, limited generalizability)</td><td align="left" valign="top">Low (clear AI use)</td><td align="left" valign="top">Low (consistent implementation)</td><td align="left" valign="top">Low (adequate data capture)</td><td align="left" valign="top">Serious (custom workload measure, not validated NASA-TLX)</td><td align="left" valign="top">Low (all outcomes reported)</td><td align="left" valign="top">Serious</td></tr><tr><td align="left" valign="top">Richardson et al (2019) [<xref ref-type="bibr" rid="ref49">49</xref>]</td><td align="left" valign="top">Low (randomized crossover simulation controls confounding)</td><td align="left" valign="top">Low (adequate sample n=32)</td><td align="left" valign="top">Low (clear CDSS<sup><xref ref-type="table-fn" rid="table3fn9">i</xref></sup> intervention)</td><td align="left" valign="top">Low (standardized protocol)</td><td align="left" valign="top">Low (complete data)</td><td align="left" valign="top">Low (full NASA-TLX, validated)</td><td align="left" valign="top">Low (all outcomes reported)</td><td align="left" valign="top">Low</td></tr><tr><td align="left" valign="top">Kandaswamy et al (2025) [<xref ref-type="bibr" rid="ref17">17</xref>]</td><td align="left" valign="top">Moderate (pre-post with some statistical adjustment)</td><td align="left" valign="top">Moderate (multisite pediatric units)</td><td align="left" valign="top">Low (clear sepsis AI exposure)</td><td align="left" valign="top">Low (structured implementation)</td><td align="left" valign="top">Low (complete outcome data)</td><td align="left" valign="top">Low (validated NASA-TLX)</td><td align="left" valign="top">Low (all outcomes reported)</td><td align="left" valign="top">Moderate</td></tr><tr><td align="left" valign="top">Sanderson et al (2023) [<xref ref-type="bibr" rid="ref50">50</xref>]</td><td align="left" valign="top">Low (randomized crossover design)</td><td align="left" valign="top">Low (adequate sample n=44, diverse participants)</td><td align="left" valign="top">Low (clear CDSS intervention)</td><td align="left" valign="top">Low (standardized trauma scenarios)</td><td align="left" valign="top">Low (complete data)</td><td align="left" valign="top">Low (full NASA-TLX)</td><td align="left" valign="top">Low (all outcomes reported)</td><td align="left" valign="top">Low</td></tr><tr><td align="left" valign="top">Garcia et al (2024) [<xref ref-type="bibr" rid="ref37">37</xref>]</td><td align="left" valign="top">Moderate (pre-post without control, novel intervention)</td><td align="left" valign="top">Moderate (single academic site, early adopters)</td><td align="left" valign="top">Low (clear LLM<sup><xref ref-type="table-fn" rid="table3fn10">j</xref></sup> inbox tool use)</td><td align="left" valign="top">Low (consistent implementation)</td><td align="left" valign="top">Moderate (30% nonresponse to follow-up survey)</td><td align="left" valign="top">Low (validated instruments)</td><td align="left" valign="top">Low (all outcomes reported)</td><td align="left" valign="top">Moderate</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>ROBINS-I: Risk of Bias in Non-randomized Studies of Interventions.</p></fn><fn id="table3fn2"><p><sup>b</sup>AI: artificial intelligence.</p></fn><fn id="table3fn3"><p><sup>c</sup>EHR: electronic health record.</p></fn><fn id="table3fn4"><p><sup>d</sup>QI: quality improvement.</p></fn><fn id="table3fn5"><p><sup>e</sup>NASA-TLX: National Aeronautics and Space Administration Task Load Index.</p></fn><fn id="table3fn6"><p><sup>f</sup>DAX: Dragon Ambient Experience.</p></fn><fn id="table3fn7"><p><sup>g</sup>SUS: System Usability Scale.</p></fn><fn id="table3fn8"><p><sup>h</sup>ITSA: interrupted time series analysis.</p></fn><fn id="table3fn9"><p><sup>i</sup>CDSS: clinical decision support system.</p></fn><fn id="table3fn10"><p><sup>j</sup>LLM: large language model.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-5"><title>Certainty of Evidence</title><p>The certainty of evidence was assessed using the GRADE approach for each outcome-intervention combination (<xref ref-type="table" rid="table4">Table 4</xref>). The certainty of evidence for cognitive workload reduction with ambient AI documentation was rated as moderate (&#x2295;&#x2295;&#x2295;&#x25EF;). Evidence from 13 studies (2 RCTs and 11 observational or quasi-experimental) [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref42">42</xref>-<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref52">52</xref>] consistently demonstrated NASA-TLX reductions of 24-40 points. The evidence was not downgraded for risk of bias, as the 2 RCTs showed low risk and observational studies showed consistent direction of effect. No serious inconsistency, indirectness, or imprecision was identified. Because every pool contained fewer than 10 studies, formal small-study-effect testing was not performed. Consequently, the possibility of small-study effects was considered qualitatively and judged not to warrant downgrading for this outcome.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>GRADE (Grading of Recommendations Assessment, Development and Evaluation) summary of findings for the 6 prespecified meta-analytic pools (uniform Knapp-Hartung adjustment, restricted maximum likelihood estimation of &#x03C4;<sup>2</sup>).</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Question</td><td align="left" valign="bottom">Studies, n</td><td align="left" valign="bottom">Patients, n</td><td align="left" valign="bottom" colspan="6">Certainty assessment</td><td align="left" valign="bottom" colspan="2">Effect</td><td align="left" valign="bottom">Certainty</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">Study design</td><td align="left" valign="top">Risk of bias</td><td align="left" valign="top">Inconsistency</td><td align="left" valign="top">Indirectness</td><td align="left" valign="top">Imprecision</td><td align="left" valign="top">Other considerations</td><td align="left" valign="top">Relative (95% CI)</td><td align="left" valign="top">Absolute (95% CI)</td><td align="left" valign="top"/></tr></thead><tbody><tr><td align="left" valign="top">Ambient AI<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup> documentation compared to control in NASA-TLX<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup> mental demand</td><td align="char" char="." valign="top">2</td><td align="char" char="." valign="top">308</td><td align="left" valign="top">Nonrandomized studies</td><td align="left" valign="top">Serious<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="top">Serious<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="left" valign="top">Not serious</td><td align="left" valign="top">Serious<sup><xref ref-type="table-fn" rid="table4fn5">e</xref></sup></td><td align="left" valign="top">None</td><td align="char" char="." valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table4fn6">f</xref></sup></td><td align="left" valign="top">SMD<sup><xref ref-type="table-fn" rid="table4fn7">g</xref></sup> 1.29 SD lower (3.64 lower to 1.07 higher)</td><td align="left" valign="top">&#x2A01;&#x25EF;&#x25EF;&#x25EF;<break/>Very low<sup><xref ref-type="table-fn" rid="table4fn1">c,d,e</xref></sup></td></tr><tr><td align="left" valign="top">Ambient AI documentation compared to control for NASA-TLX temporal demand</td><td align="char" char="." valign="top">2</td><td align="char" char="." valign="top">308</td><td align="left" valign="top">Nonrandomized studies</td><td align="left" valign="top">Serious<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="top">Not serious</td><td align="left" valign="top">Not serious</td><td align="left" valign="top">Serious<sup><xref ref-type="table-fn" rid="table4fn5">e</xref></sup></td><td align="left" valign="top">None</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">SMD 1.46 SD lower (2.81 lower to 0.11 lower)</td><td align="left" valign="top">&#x2A01;&#x2A01;&#x25EF;&#x25EF;<break/>Low<sup><xref ref-type="table-fn" rid="table4fn1">c,e</xref></sup></td></tr><tr><td align="left" valign="top">Ambient AI documentation compared to control for NASA-TLX effort</td><td align="char" char="." valign="top">2</td><td align="char" char="." valign="top">308</td><td align="left" valign="top">Nonrandomized studies</td><td align="left" valign="top">Serious<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="top">Not serious</td><td align="left" valign="top">Not serious</td><td align="left" valign="top">Not serious<sup><xref ref-type="table-fn" rid="table4fn8">h</xref></sup></td><td align="left" valign="top">None</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">SMD 1.29 SD lower (2.16 lower to 0.42 lower)</td><td align="left" valign="top">&#x2A01;&#x2A01;&#x2A01;&#x25EF;<break/>Moderate<sup><xref ref-type="table-fn" rid="table4fn1">c,h</xref></sup></td></tr><tr><td align="left" valign="top">Ambient AI documentation compared to control for PFI<sup><xref ref-type="table-fn" rid="table4fn9">i</xref></sup> work exhaustion</td><td align="char" char="." valign="top">3</td><td align="char" char="." valign="top">375</td><td align="left" valign="top">Nonrandomized studies</td><td align="left" valign="top">Serious<sup><xref ref-type="table-fn" rid="table4fn10">j</xref></sup></td><td align="left" valign="top">Not serious</td><td align="left" valign="top">Not serious</td><td align="left" valign="top">Serious<sup><xref ref-type="table-fn" rid="table4fn11">k</xref></sup></td><td align="left" valign="top">None</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">MD<sup><xref ref-type="table-fn" rid="table4fn12">l</xref></sup> 0.35 lower (0.58 lower to 0.12 lower)</td><td align="left" valign="top">&#x2A01;&#x2A01;&#x25EF;&#x25EF;<break/>Low<sup><xref ref-type="table-fn" rid="table4fn1">j,k</xref></sup></td></tr><tr><td align="left" valign="top">Ambient AI documentation compared to control for burnout prevalence</td><td align="char" char="." valign="top">3</td><td align="char" char="." valign="top">502</td><td align="left" valign="top">Nonrandomized studies</td><td align="left" valign="top">Serious<sup><xref ref-type="table-fn" rid="table4fn10">j</xref></sup></td><td align="left" valign="top">Not serious</td><td align="left" valign="top">Not serious</td><td align="left" valign="top">Not serious</td><td align="left" valign="top">Publication bias strongly suspected<sup><xref ref-type="table-fn" rid="table4fn13">m</xref></sup></td><td align="left" valign="top">OR<sup><xref ref-type="table-fn" rid="table4fn14">n</xref></sup> 0.47 (0.25 to 0.86)</td><td align="left" valign="top">180 fewer per 1000 (from 300 fewer to 38 fewer)</td><td align="left" valign="top">&#x2A01;&#x2A01;&#x25EF;&#x25EF;<break/>Low<sup><xref ref-type="table-fn" rid="table4fn1">j,m</xref></sup></td></tr><tr><td align="left" valign="top">Ambient AI documentation compared to control for documentation time</td><td align="char" char="." valign="top">2</td><td align="char" char="." valign="top">137</td><td align="left" valign="top">Nonrandomized studies</td><td align="left" valign="top">Serious<sup><xref ref-type="table-fn" rid="table4fn15">o</xref></sup></td><td align="left" valign="top">Not serious</td><td align="left" valign="top">Not serious</td><td align="left" valign="top">Serious<sup><xref ref-type="table-fn" rid="table4fn5">e</xref></sup></td><td align="left" valign="top">None</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">SMD 0.24 SD lower (1.1 lower to 0.61 higher)</td><td align="left" valign="top">&#x2A01;&#x2A01;&#x25EF;&#x25EF;<break/>Low<sup><xref ref-type="table-fn" rid="table4fn1">e,o</xref></sup></td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>AI: artificial intelligence.</p></fn><fn id="table4fn2"><p><sup>b</sup>NASA-TLX: National Aeronautics and Space Administration Task Load Index.</p></fn><fn id="table4fn3"><p><sup>c</sup>Observational pre-post designs.</p></fn><fn id="table4fn4"><p><sup>d</sup>High <italic>I</italic><sup>2</sup>.</p></fn><fn id="table4fn5"><p><sup>e</sup>CI crosses null at k=2 under Knapp-Hartung.</p></fn><fn id="table4fn6"><p><sup>f</sup>Not available.</p></fn><fn id="table4fn7"><p><sup>g</sup>SMD: standardized mean difference.</p></fn><fn id="table4fn8"><p><sup>h</sup>k=2.</p></fn><fn id="table4fn9"><p><sup>i</sup>PFI: Professional Fulfillment Index.</p></fn><fn id="table4fn10"><p><sup>j</sup>Predominantly observational pre-post designs.</p></fn><fn id="table4fn11"><p><sup>k</sup>95% prediction interval crosses null.</p></fn><fn id="table4fn12"><p><sup>l</sup>MD: mean difference.</p></fn><fn id="table4fn13"><p><sup>m</sup>Possible small-study effects/publication bias (small literature concentrated on two commercial products).</p></fn><fn id="table4fn14"><p><sup>n</sup>OR: odds ratio.</p></fn><fn id="table4fn15"><p><sup>o</sup>Observational designs.</p></fn></table-wrap-foot></table-wrap><p>The certainty for burnout reduction with ambient AI documentation was rated as low (&#x2295;&#x2295;&#x25EF;&#x25EF;). In total, 10 studies (2 RCTs and 8 observational) [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref45">45</xref>] reported burnout prevalence reductions of 7&#x2010;26 percentage points. Evidence was not downgraded for risk of bias, given RCT support. However, we noted possible small-study effects, given a small literature concentrated on 2 commercial products in which negative or null studies may be underrepresented [<xref ref-type="bibr" rid="ref40">40</xref>].</p><p>The certainty for documentation time reduction with ambient AI documentation was rated as low (&#x2295;&#x2295;&#x25EF;&#x25EF;). In total, 8 studies (2 RCTs and 6 observational or quasi-experimental) [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref46">46</xref>] reported time savings of 0.9&#x2010;3 minutes per appointment. Evidence was downgraded for serious risk of bias (predominantly observational designs) and serious inconsistency, as effect sizes varied substantially across studies and settings.</p><p>The certainty regarding the effects of diagnostic imaging AI on cognitive workload was rated as very low (&#x2295;&#x25EF;&#x25EF;&#x25EF;). In total, 3 observational studies [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref48">48</xref>] showed inconsistent results ranging from no significant reduction to increased workload. Evidence was downgraded for serious risk of bias (all observational), serious inconsistency (conflicting directions of effect), and serious imprecision due to small sample sizes.</p><p>Similarly, the certainty for CDSS effects on cognitive workload was rated as very low (&#x2295;&#x25EF;&#x25EF;&#x25EF;). In total, 3 observational studies [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref49">49</xref>,<xref ref-type="bibr" rid="ref50">50</xref>] showed highly variable results, with workload reductions of 7%&#x2010;50% in some studies but increases of 14 points in others (sepsis AI). Evidence was downgraded for very serious inconsistency (contradictory findings), serious indirectness (heterogeneous CDSS types and clinical contexts), and serious imprecision.</p><p>The certainty for AI-based burnout intervention was rated as low (&#x2295;&#x2295;&#x25EF;&#x25EF;). One RCT [<xref ref-type="bibr" rid="ref51">51</xref>] demonstrated CBI reductions of 20&#x2010;23 points, but evidence was downgraded for serious imprecision due to reliance on a single study.</p><p>Taken together, the certainty of evidence for ambient AI documentation was limited both because the literature is commercially concentrated (Abridge and DAX Copilot dominate) and because the 95% PI for both burnout pools crosses the null effect line. Evidence for diagnostic imaging AI and CDSS remains very low certainty, with unintended workload increases observed in single studies of CADe imaging [<xref ref-type="bibr" rid="ref16">16</xref>] and alert-based sepsis CDSS [<xref ref-type="bibr" rid="ref17">17</xref>]. Documentation time benefits among ambient AI adopters carry low certainty (SMD &#x2212;0.24, 95% CI &#x2212;1.10 to 0.61).</p></sec><sec id="s3-6"><title>Synthesis of Findings</title><sec id="s3-6-1"><title>Ambient AI Documentation Systems</title><p>In total, 13 studies examined cognitive workload and burnout outcomes associated with ambient AI documentation tools [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref42">42</xref>-<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref52">52</xref>], representing the largest evidence base in this review. These AI scribe systems convert patient-clinician conversations into draft clinical notes using speech recognition and natural language processing. The highest-quality evidence came from 2 RCTs. Lukac et al [<xref ref-type="bibr" rid="ref7">7</xref>] conducted a 3-arm pragmatic RCT comparing 2 ambient AI scribes (DAX Copilot; Nuance Communications and Nabla; Nabla SAS) against usual care in 238 outpatient physicians across 14 specialties. Physicians randomized to AI scribes demonstrated significant reductions in task load, with DAX Copilot showing a 39.9-point reduction and Nabla showing a 31.7-point reduction on the Physician Task Load scale (<italic>P</italic>&#x003C;.01). Work exhaustion also decreased significantly (&#x2212;0.27 on a 0&#x2010;4 scale; <italic>P</italic>=.01). Documentation time decreased by 9.5% with Nabla (<italic>P</italic>=.02) but showed no significant change with DAX Copilot. Afshar et al [<xref ref-type="bibr" rid="ref8">8</xref>] used a 24-week stepped-wedge individually randomized design with 66 health care practitioners, finding that Abridge (Abridge Inc) significantly reduced the work exhaustion and interpersonal disengagement composite score by 0.44 points (95% CI &#x2212;0.62 to &#x2212;0.25; <italic>P</italic>&#x003C;.001) on the Stanford PFI.</p><p>Pre- and postimplementation studies demonstrated consistent findings. Olson et al [<xref ref-type="bibr" rid="ref4">4</xref>] reported that burnout prevalence decreased from 51.9% to 38.8% (adjusted OR 0.26, 95% CI 0.13&#x2010;0.54; <italic>P</italic>&#x003C;.001) and cognitive task load decreased by 2.64 points (<italic>P</italic>&#x003C;.001) among 263 clinicians following Abridge implementation across 6 health systems. Shah et al [<xref ref-type="bibr" rid="ref42">42</xref>] found that NASA-TLX scores decreased by 24.42 points (<italic>P</italic>&#x003C;.001) and Stanford PFI burnout scores decreased by 1.94 points (<italic>P</italic>&#x003C;.001) among 48 physicians using DAX Copilot. Hudson et al [<xref ref-type="bibr" rid="ref43">43</xref>] demonstrated a 46.6% reduction in NASA-TLX composite scores (221.2 to 118.2; <italic>P</italic>&#x003C;.001) in a randomized crossover study of 40 providers, with significant reductions across all subscales including mental demand (48.8% reduction), temporal demand (44.4% reduction), and effort (46.3% reduction).</p><p>The largest observational study by You et al [<xref ref-type="bibr" rid="ref9">9</xref>] enrolled 1430 physicians and APPs across 2 academic medical centers; among the 265 Mass General Brigham clinicians who completed paired surveys at 42 days, burnout prevalence fell by 21.2 percentage points (50.6% to 29.4%; <italic>&#x03C7;</italic><sup>2</sup><sub>1</sub>=42.4; <italic>P</italic>&#x003C;.001), and well-being rose by 30.7 percentage points following ambient AI implementation over 42 days. Owens et al [<xref ref-type="bibr" rid="ref44">44</xref>] found significantly lower OLBI disengagement scores in high versus low DAX users (16.3 vs 18.4; difference &#x2212;2.1, 95% CI &#x2212;3.8 to &#x2212;0.4) among 110 primary care providers. Misurac et al [<xref ref-type="bibr" rid="ref45">45</xref>] reported that burnout prevalence decreased from 69% to 43% (<italic>P</italic>=.005) among 35 providers. Stults et al [<xref ref-type="bibr" rid="ref10">10</xref>] reported that documentation time decreased from 6.2 to 5.3 minutes per appointment (<italic>P</italic>&#x003C;.001) and burnout prevalence from 42.1% to 35.1% among 100 ambulatory clinicians at Sutter Health using Abridge over 12 weeks. In this cohort, NASA-TLX mental, temporal, and effort subscales all decreased significantly. Duggan et al [<xref ref-type="bibr" rid="ref39">39</xref>] found positive System Usability Scale scores and perceived reduction in documentation burden among 46 physicians and APPs using DAX Copilot over 8 weeks, though individual experiences were mixed.</p><p>Pelletier et al [<xref ref-type="bibr" rid="ref38">38</xref>] provided the first pediatric ambient AI evidence. Documentation time decreased by 2.8 minutes per appointment (<italic>P</italic>&#x003C;.001) with 1.5 hours saved weekly, alongside significant reductions in NASA-TLX workload and Mini-Z burnout scores among 84 pediatric physicians and APPs at Akron Children&#x2019;s Hospital using Abridge over 6 months. The interrupted time series analysis design strengthened causal inference. A small simulation study by Bracken et al [<xref ref-type="bibr" rid="ref46">46</xref>] examined Heidi Health (Heidi Health Pty Ltd) among 7 junior doctors in a simulated inpatient setting. The authors reported substantial reductions in documentation time (27 vs 128 seconds for progress notes; <italic>P</italic>&#x003C;.0001) and NASA-TLX subscale reductions including a 79% reduction in frustration and an 81% reduction in effort, though the small sample size and simulation design limit generalizability. Similarly, Nawaz et al [<xref ref-type="bibr" rid="ref52">52</xref>] conducted a crossover simulation study evaluating Lyrebird Health among 8 psychiatrists, demonstrating the largest effect size observed in this review, with NASA-TLX total workload scores of 25.0 with AI versus 461.3 without AI (MD &#x2212;436.25; <italic>P</italic>&#x003C;.001). Documentation quality also improved significantly on the Sheffield Assessment Instrument for Letters. However, the preprint status, small sample size (n=8), and simulation design warrant cautious interpretation.</p><p>The categories of diagnostic imaging AI, CDSS, and LLM-based tools were not eligible for meta-analytic pooling because of heterogeneity in study design, AI subtype, and outcome instrumentation. Study-level characteristics and key findings for these categories are reported in <xref ref-type="table" rid="table1">Tables 1</xref>, <xref ref-type="table" rid="table3">3</xref>, and <xref ref-type="table" rid="table4">4</xref>. Category-level interpretation is provided in the Discussion section.</p></sec><sec id="s3-6-2"><title>AI-Based Burnout Intervention</title><p>One study examined AI not as a clinical tool but as a mechanism for delivering personalized burnout interventions. Baek and Cha [<xref ref-type="bibr" rid="ref51">51</xref>] conducted a single-blind 3-group RCT among 120 nurses, comparing AI-tailored burnout interventions through a mobile app (Nurse Healing Space; Ewha Womans University) against standardized interventions and a waitlist control. The study achieved 0 dropouts. The AI-tailored group demonstrated significant reductions in CBI scores for client-related burnout (62.6 to 42.0; <italic>F</italic>=7.73; <italic>P</italic>=.001) and personal burnout (67.5 to 44.7; <italic>F</italic>=10.97; <italic>P</italic>&#x003C;.0001) compared to controls. This study represents a distinct application of AI&#x2014;using algorithmic personalization to address rather than potentially contribute to clinician burden.</p></sec><sec id="s3-6-3"><title>Meta-Analysis Results</title><p>Meta-analysis was conducted for ambient AI documentation studies with poolable data. All 6 prespecified pools were synthesized under the HKSJ adjustment with restricted maximum likelihood estimation of &#x03C4;<sup>2</sup> as the primary inferential framework, irrespective of k (with <italic>q</italic>* truncated to 1 per IntHout et al [<xref ref-type="bibr" rid="ref34">34</xref>]). Forest plots for each pool are presented as <xref ref-type="fig" rid="figure3">Figures 3</xref><xref ref-type="fig" rid="figure4"/><xref ref-type="fig" rid="figure5"/><xref ref-type="fig" rid="figure6"/><xref ref-type="fig" rid="figure7"/>-<xref ref-type="fig" rid="figure8">8</xref>.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Forest plot of NASA-TLX mental demand subscale for ambient AI documentation versus baseline (k=2) [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. Pooled SMD under the Knapp-Hartung-Sidik-Jonkman adjustment with REML estimation of &#x03C4;<sup>2</sup> (<italic>q</italic>* truncated to 1) was SMD &#x2212;1.29 (95% CI &#x2212;3.64 to 1.07); <italic>I</italic><sup>2</sup>=75.3%. AI: artificial intelligence; HKSJ: Hartung-Knapp-Sidik-Jonkman; NASA-TLX: National Aeronautics and Space Administration Task Load Index; RE: random effect; REML: restricted maximum likelihood; SMD: standardized mean difference.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e93618_fig03.png"/></fig><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Forest plot of NASA-TLX temporal demand subscale for ambient AI documentation versus baseline (k=2) [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. Knapp-Hartung-Sidik-Jonkman&#x2013;adjusted pooled SMD &#x2212;1.46 (95% CI &#x2212;2.81 to &#x2212;0.11); <italic>I</italic><sup>2</sup>=31.1%. AI: artificial intelligence; HKSJ: Hartung-Knapp-Sidik-Jonkman; NASA-TLX: National Aeronautics and Space Administration Task Load Index; RE: random effect; SMD: standardized mean difference.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e93618_fig04.png"/></fig><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Forest plot of NASA-TLX effort subscale for ambient AI documentation versus baseline (k=2) [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. Knapp-Hartung-Sidik-Jonkman&#x2013;adjusted pooled SMD &#x2212;1.29 (95% CI &#x2212;2.16 to &#x2212;0.42); <italic>I</italic><sup>2</sup>=0%. AI: artificial intelligence; HKSJ: Hartung-Knapp-Sidik-Jonkman; NASA-TLX: National Aeronautics and Space Administration Task Load Index; RE: random effect; SMD: standardized mean difference.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e93618_fig05.png"/></fig><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Forest plot of Professional Fulfillment Index work-exhaustion subscale for ambient AI documentation versus baseline (k=3; n=375) [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref37">37</xref>]. Knapp-Hartung-Sidik-Jonkman-adjusted pooled mean difference &#x2212;0.35 (95% CI &#x2212;0.58 to &#x2212;0.12); <italic>I</italic><sup>2</sup>=0%; 95% PI &#x2212;1.03 to 0.33 (crosses null). AI: artificial intelligence; HKSJ: Hartung-Knapp-Sidik-Jonkman; MD: mean difference; PI: prediction interval; RE: random effect.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e93618_fig06.png"/></fig><fig position="float" id="figure7"><label>Figure 7.</label><caption><p>Forest plot of burnout prevalence (single-item or validated burnout instrument) for ambient AI documentation versus baseline (k=3; n=502) [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref38">38</xref>]. Knapp-Hartung-Sidik-Jonkman&#x2013;adjusted pooled OR 0.47 (95% CI 0.25-0.86); <italic>I</italic><sup>2</sup>=0%; 95% PI 0.06-3.82 (crosses null). AI: artificial intelligence; HKSJ: Hartung-Knapp-Sidik-Jonkman; OR: odds ratio; PI: prediction interval; RE: random effect.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e93618_fig07.png"/></fig><fig position="float" id="figure8"><label>Figure 8.</label><caption><p>Forest plot of documentation time for ambient AI documentation versus baseline (k=2; n=137) [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref39">39</xref>]. Knapp-Hartung-Sidik-Jonkman&#x2013;adjusted pooled SMD &#x2212;0.24 (95% CI &#x2212;1.10 to 0.61); <italic>I</italic><sup>2</sup>=0%. AI: artificial intelligence; HKSJ: Hartung-Knapp-Sidik-Jonkman; RE: random effect; SMD: standardized mean difference.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e93618_fig08.png"/></fig><p>For NASA-TLX subscales (2 studies; n=305&#x2010;311) [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref10">10</xref>], SMDs were calculated due to different measurement scales (0&#x2010;20 vs 0&#x2010;10). When applying uniform HKSJ pooling at k=2, the subscales for mental and temporal demand yielded divergent findings regarding statistical significance. Mental demand did not achieve significance (SMD &#x2212;1.291, 95% CI &#x2212;3.645 to 1.068; <italic>I</italic><sup>2</sup>=75.3%), whereas temporal demand demonstrated a significant reduction, as its CI excluded the null (SMD &#x2212;1.458, 95% CI &#x2212;2.808 to &#x2212;0.109; <italic>I</italic><sup>2</sup>=31.1%). The wide CI observed for mental demand reflects a combination of the conservative <italic>t</italic>-critical value at 1 degree of freedom and nontrivial between-study variance. Effort retained statistical significance with no detectable heterogeneity (SMD &#x2212;1.291, 95% CI &#x2212;2.160 to &#x2212;0.421; <italic>I</italic><sup>2</sup>=0%; <italic>q</italic>* truncated to 1). PIs were not estimable at k=2 (<xref ref-type="fig" rid="figure3">Figures 3-5</xref>).</p><p>For PFI work exhaustion (3 studies; n=375; [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref37">37</xref>]), the HKSJ-adjusted pooled MD was &#x2212;0.350 (95% CI &#x2212;0.582 to &#x2212;0.119; <italic>I</italic><sup>2</sup>=0%; &#x03C4;<sup>2</sup>=0); the 95% PI was &#x2212;1.034 to 0.333 and crossed the null effect line (<xref ref-type="fig" rid="figure6">Figure 6</xref>).</p><p>For burnout prevalence (3 studies; n=502; [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref38">38</xref>]), ambient AI implementation was associated with reduced odds of burnout: pooled OR was 0.470 (95% CI 0.254-0.861; <italic>I</italic><sup>2</sup>=0%; &#x03C4;<sup>2</sup>=0); the 95% PI was 0.059-3.817 and crossed the null effect line (<xref ref-type="fig" rid="figure7">Figure 7</xref>). You et al [<xref ref-type="bibr" rid="ref9">9</xref>] contributed the largest burnout sample to this pool (n=265). Leave-one-out exclusion of You et al [<xref ref-type="bibr" rid="ref9">9</xref>] preserved the direction of effect (Olson et al [<xref ref-type="bibr" rid="ref4">4</xref>] showed a 13.1 percentage-point absolute reduction in burnout prevalence [51.9% &#x2192; 38.8%]; Pelletier et al [<xref ref-type="bibr" rid="ref38">38</xref>] showed a 21.6 percentage-point reduction [54.9% &#x2192; 33.3%]), though pooled precision was reduced because of the resulting k=2.</p><p>For documentation time (2 studies; n=137) [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref39">39</xref>], the HKSJ-adjusted pooled SMD was &#x2212;0.243 (95% CI &#x2212;1.096 to 0.609; <italic>I</italic><sup>2</sup>=0%; <italic>q</italic>* truncated to 1). The effect direction favored ambient AI but did not reach statistical significance under conservative pooling at k=2 (<xref ref-type="fig" rid="figure8">Figure 8</xref>). Sensitivity analyses varying the assumed pre-post correlation (<italic>r</italic>=0.5, 0.7, and 0.9) demonstrated stable point estimates across all outcomes; the width of CIs at k=2 was driven primarily by <italic>t</italic>-critical inflation rather than by the correlation assumption (<xref ref-type="table" rid="table5">Tables 5</xref> and <xref ref-type="table" rid="table6">6</xref>).</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Sensitivity of pooled effect estimates to the assumed pre-post correlation (<italic>r</italic>=0.5, 0.7, and 0.9) under uniform Hartung-Knapp-Sidik-Jonkman adjustment with restricted maximum likelihood estimation of &#x03C4;<sup>2</sup><sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup>.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Outcome (k studies) and <italic>r</italic></td><td align="left" valign="bottom">Effect size (95% CI)</td><td align="left" valign="bottom"><italic>I</italic><sup>2</sup> (%)</td><td align="left" valign="bottom">&#x03C4;<sup>2<xref ref-type="table-fn" rid="table5fn2">b</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top" colspan="4">NASA-TLX<sup><xref ref-type="table-fn" rid="table5fn3">c</xref></sup> mental demand (k=2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>0.5</td><td align="left" valign="top">&#x2212;1.28 (&#x2212;3.62 to 1.06)</td><td align="left" valign="top">69.4</td><td align="left" valign="top">0.049</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>0.7<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup></td><td align="left" valign="top">&#x2212;1.29 (&#x2212;3.64 to 1.07)</td><td align="left" valign="top">75.3</td><td align="left" valign="top">0.053</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>0.9</td><td align="left" valign="top">&#x2212;1.30 (&#x2212;3.67 to 1.08)</td><td align="left" valign="top">81.2</td><td align="left" valign="top">0.058</td></tr><tr><td align="left" valign="top" colspan="4">NASA-TLX temporal demand (k=2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>0.5</td><td align="left" valign="top">&#x2212;1.45 (&#x2212;2.71 to &#x2212;0.18)</td><td align="left" valign="top">16.6</td><td align="left" valign="top">0.005</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>0.7<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup></td><td align="left" valign="top">&#x2212;1.46 (&#x2212;2.81 to &#x2212;0.11)</td><td align="left" valign="top">31.1</td><td align="left" valign="top">0.009</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>0.9</td><td align="left" valign="top">&#x2212;1.47 (&#x2212;2.88 to &#x2212;0.05)</td><td align="left" valign="top">45.6</td><td align="left" valign="top">0.013</td></tr><tr><td align="left" valign="top" colspan="4">NASA-TLX effort (k=2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>0.5</td><td align="left" valign="top">&#x2212;1.29 (&#x2212;2.27 to &#x2212;0.31)</td><td align="left" valign="top">0.0</td><td align="left" valign="top">0.000</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>0.7<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup></td><td align="left" valign="top">&#x2212;1.29 (&#x2212;2.16 to &#x2212;0.42)</td><td align="left" valign="top">0.0</td><td align="left" valign="top">0.000</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>0.9</td><td align="left" valign="top">&#x2212;1.29 (&#x2212;2.03 to &#x2212;0.55)</td><td align="left" valign="top">0.0</td><td align="left" valign="top">0.000</td></tr><tr><td align="left" valign="top" colspan="4">Documentation time (k=2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>0.5</td><td align="left" valign="top">&#x2212;0.24 (&#x2212;1.33 to 0.85)</td><td align="left" valign="top">0.0</td><td align="left" valign="top">0.000</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>0.7<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup></td><td align="left" valign="top">&#x2212;0.24 (&#x2212;1.10 to 0.61)</td><td align="left" valign="top">0.0</td><td align="left" valign="top">0.000</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>0.9</td><td align="left" valign="top">&#x2212;0.24 (&#x2212;0.76 to 0.27)</td><td align="left" valign="top">0.0</td><td align="left" valign="top">0.000</td></tr><tr><td align="left" valign="top" colspan="4">Professional Fulfillment Index work exhaustion (k=3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>N/A<sup><xref ref-type="table-fn" rid="table5fn5">e</xref></sup></td><td align="left" valign="top">&#x2212;0.35<sup><xref ref-type="table-fn" rid="table5fn6">f</xref></sup> (&#x2212;0.58 to &#x2212;0.12)</td><td align="left" valign="top">0.0</td><td align="left" valign="top">0.000</td></tr><tr><td align="left" valign="top" colspan="4">Burnout prevalence (k=3)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>0.5<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup></td><td align="left" valign="top">0.47<sup><xref ref-type="table-fn" rid="table5fn7">g</xref></sup> (0.25 to 0.86)</td><td align="left" valign="top">0.0</td><td align="left" valign="top">0.007</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>0.7</td><td align="left" valign="top">0.47<sup><xref ref-type="table-fn" rid="table5fn7">g</xref></sup> (0.25 to 0.86)</td><td align="left" valign="top">0.0</td><td align="left" valign="top">0.007</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>0.9</td><td align="left" valign="top">0.47<sup><xref ref-type="table-fn" rid="table5fn7">g</xref></sup> (0.25 to 0.86)</td><td align="left" valign="top">0.0</td><td align="left" valign="top">0.007</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup><italic>q</italic>* was truncated to a minimum of 1 at k=2 [<xref ref-type="bibr" rid="ref34">34</xref>]. For binary outcomes (burnout prevalence), log-odds ratios were computed directly from 2&#x00D7;2 tables; the pre-post correlation parameter is conventional and does not enter the variance calculation.</p></fn><fn id="table5fn2"><p><sup>b</sup>&#x03C4;<sup>2</sup>: between-study variance.</p></fn><fn id="table5fn3"><p><sup>c</sup>NASA-TLX: National Aeronautics and Space Administration Task Load Index.</p></fn><fn id="table5fn4"><p><sup>d</sup>Primary analysis.</p></fn><fn id="table5fn5"><p><sup>e</sup>Not applicable. The work exhaustion pool was pooled using mean differences reported directly by the primary studies, so the assumed pre-post correlation did not enter the variance calculation.</p></fn><fn id="table5fn6"><p><sup>f</sup>Mean difference.</p></fn><fn id="table5fn7"><p><sup>g</sup>Odds ratio.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Comparison of conventional random-effects pooling (<italic>z</italic>-based) versus uniform Hartung-Knapp-Sidik-Jonkman (HKSJ) pooling (<italic>t</italic>-based with <italic>q</italic>* truncated to 1) for the 4 k=2 prespecified pools (assumed pre-post correlation <italic>r</italic>=0.7)<sup><xref ref-type="table-fn" rid="table6fn1">a</xref></sup>.</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Outcome (k=2) and method</td><td align="left" valign="bottom">SMD<sup><xref ref-type="table-fn" rid="table6fn2">b</xref></sup> (95% CI)</td><td align="left" valign="bottom">CI width<sup><xref ref-type="table-fn" rid="table6fn3">c</xref></sup></td><td align="left" valign="bottom">Ratio<sup><xref ref-type="table-fn" rid="table6fn4">d</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top" colspan="4">NASA-TLX<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup> mental demand (k=2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RE<sup><xref ref-type="table-fn" rid="table6fn6">f</xref></sup> (<italic>z</italic>-based, no HKSJ)</td><td align="left" valign="top">&#x2212;1.29 (&#x2212;1.65 to &#x2212;0.93)</td><td align="left" valign="top">0.73</td><td align="left" valign="top">1.0 (reference)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RE+uniform HKSJ</td><td align="left" valign="top">&#x2212;1.29 (&#x2212;3.64 to 1.07)</td><td align="left" valign="top">4.71</td><td align="left" valign="top">6.5&#x00D7;</td></tr><tr><td align="left" valign="top" colspan="4">NASA-TLX temporal demand (k=2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RE (<italic>z</italic>-based, no HKSJ)</td><td align="left" valign="top">&#x2212;1.46 (&#x2212;1.67 to &#x2212;1.25)</td><td align="left" valign="top">0.42</td><td align="left" valign="top">1.0 (reference)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RE+uniform HKSJ</td><td align="left" valign="top">&#x2212;1.46 (&#x2212;2.81 to &#x2212;0.11)</td><td align="left" valign="top">2.70</td><td align="left" valign="top">6.5&#x00D7;</td></tr><tr><td align="left" valign="top" colspan="4">NASA-TLX effort (k=2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RE (<italic>z</italic>-based, no HKSJ)</td><td align="left" valign="top">&#x2212;1.29 (&#x2212;1.42 to &#x2212;1.16)</td><td align="left" valign="top">0.27</td><td align="left" valign="top">1.0 (reference)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RE+uniform HKSJ</td><td align="left" valign="top">&#x2212;1.29 (&#x2212;2.16 to &#x2212;0.42)</td><td align="left" valign="top">1.74</td><td align="left" valign="top">6.5&#x00D7;</td></tr><tr><td align="left" valign="top" colspan="4">Documentation time (k=2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RE (<italic>z</italic>-based, no HKSJ)</td><td align="left" valign="top">&#x2212;0.24 (&#x2212;0.37 to &#x2212;0.11)</td><td align="left" valign="top">0.26</td><td align="left" valign="top">1.0 (reference)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RE+uniform HKSJ</td><td align="left" valign="top">&#x2212;0.24 (&#x2212;1.10 to 0.61)</td><td align="left" valign="top">1.70</td><td align="left" valign="top">6.5&#x00D7;</td></tr></tbody></table><table-wrap-foot><fn id="table6fn1"><p><sup>a</sup>The width of the 95% CI under uniform HKSJ exceeds the conventional <italic>z</italic>-based interval by a factor of approximately 6.5 for all 4 k=2 pools; this reflects the <italic>t</italic>-critical value at 1 degree of freedom (<italic>t</italic><sub>&#x2080;.&#x2080;&#x2082;&#x2085;, &#x2081;</sub>=12.71) relative to the standard-normal critical value (<italic>z</italic><sub>&#x2080;.&#x2080;&#x2082;&#x2085;</sub>=1.96). Point estimates are unchanged. This demonstrates that for the k=2 outcomes that did not reach significance under uniform HKSJ pooling, the width of the resulting CI was driven primarily by <italic>t</italic>-critical inflation rather than by between-study heterogeneity or the assumed pre-post correlation. </p></fn><fn id="table6fn2"><p><sup>b</sup>SMD: standardized mean difference.</p></fn><fn id="table6fn3"><p><sup>c</sup>CI width is computed as (upper bound&#x2212;lower bound).</p></fn><fn id="table6fn4"><p><sup>d</sup>Ratio is HKSJ CI width/<italic>z</italic>-based CI width; the constant factor of approximately 6.5 corresponds to <italic>t</italic><sub>&#x2080;.&#x2080;&#x2082;&#x2085;, &#x2081;</sub>/z<sub>&#x2080;.&#x2080;&#x2082;&#x2085;</sub>=12.706/1.960.</p></fn><fn id="table6fn5"><p><sup>e</sup>NASA-TLX: National Aeronautics and Space Administration Task Load Index.</p></fn><fn id="table6fn6"><p><sup>f</sup>RE: random effect.</p></fn></table-wrap-foot></table-wrap><p>We identified 3 key interpretive points regarding the statistical pooling. First, the uniform application of the HKSJ adjustment prioritizes protection against type I error inflation over narrow-band precision, particularly when unmodeled between-study variance is present [<xref ref-type="bibr" rid="ref34">34</xref>]. Consequently, the pools for NASA-TLX mental demand and documentation time did not reach conventional statistical significance, despite their point estimates favoring ambient AI. Second, the 95% PIs for the 2 pools containing 3 studies (PFI work exhaustion and burnout prevalence) crossed the null effect line. This indicates that the true effect in a new, comparable setting could plausibly include no clinical benefit. Third, the NASA-TLX effort subscale was the only pool with 2 studies that retained statistical significance. This occurred because the estimated between-study variance was 0, which allowed the variance-inflation factor to be truncated to 1. Together, these findings reinforce the interpretation that the current pooled evidence base should be presented as preliminary rather than definitive, despite being suggestive of reductions in cognitive workload and burnout [<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref40">40</xref>].</p></sec><sec id="s3-6-4"><title>Summary of Effect Sizes</title><p>Across studies using the NASA-TLX or derivative instruments, effect sizes for AI documentation tools ranged from 14 to 40 points on the 100-point scale, representing moderate to large effects [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref46">46</xref>]. In total, 2 simulation studies reported substantially larger effects. Bracken et al [<xref ref-type="bibr" rid="ref46">46</xref>] demonstrated 79%&#x2010;81% reductions in frustration and effort subscales, and Nawaz et al [<xref ref-type="bibr" rid="ref52">52</xref>] reported an MD of 436.3 points on raw NASA-TLX total scores (25.0 vs 461.3; <italic>P</italic>&#x003C;.001). Although this represents the largest effect size observed in this review, these findings require cautious interpretation, given the controlled simulation conditions and small sample sizes.</p><p>Burnout prevalence reductions ranged from 7 to 26 percentage points in absolute terms [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref45">45</xref>]. The randomized crossover study by Hudson et al [<xref ref-type="bibr" rid="ref43">43</xref>] demonstrated the largest effect among real-world clinical studies with a 60.7% reduction in composite workload scores. In contrast, CDSS and diagnostic imaging AI showed smaller, inconsistent, or negative effects on cognitive workload [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref47">47</xref>]. Kandaswamy et al [<xref ref-type="bibr" rid="ref17">17</xref>] found workload increased by 14 points with sepsis AI implementation, and Wenderott et al [<xref ref-type="bibr" rid="ref16">16</xref>] found no workload reduction with prostate MRI AI despite increased reading time for complex cases.</p><p>Across the ambient AI documentation studies, the direction of effect was consistent. These tools were associated with reductions in cognitive workload and burnout, with the strongest evidence from the 2 RCTs [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. Multisite studies were concordant with these findings across settings including pediatric [<xref ref-type="bibr" rid="ref38">38</xref>] and psychiatric [<xref ref-type="bibr" rid="ref52">52</xref>] specialties. However, diagnostic and alerting AI tools showed mixed effects highly dependent on implementation characteristics, alert frequency, case complexity, and workflow integration. Individual response variability was noted, with Duggan et al [<xref ref-type="bibr" rid="ref39">39</xref>] reporting that not all clinicians benefited equally from ambient AI scribes. These findings suggest that the cognitive impact of AI varies substantially by application type, with documentation AI providing more consistent benefits than diagnostic or alerting AI systems.</p></sec></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This review set out to answer 3 questions, and our main findings map onto each. First, we asked how much AI changes the mental effort and strain that clinicians experience when it is built into their work, measuring this with well-established questionnaires. Combining the available studies, we found that AI used to automatically draft clinical notes (&#x201C;ambient AI documentation&#x201D;) was generally linked to less mental strain and less burnout. Of the 6 outcomes we evaluated, 4 showed a clear benefit. Two additional outcomes, mental demand and documentation time, also indicated potential benefits but lacked statistical conclusiveness due to the limited number of studies available for pooling. Second, we asked whether the effect depends on the type of AI, and it clearly did. Note-drafting AI tended to reduce workload and burnout, whereas AI systems designed to interpret medical images or issue decision-support alerts produced mixed results and, in some cases, unintended increases in clinician workload [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. Third, we asked about subtler costs, including the effort required to double-check AI output and the tendency to overtrust it. None of the 21 studies measured these constructs directly with validated tools. We therefore discuss them as ideas to guide future research rather than as measured results.</p><p>This review brings validated workload questionnaires together with a conservative pooling method and PIs, which estimate how the effect might vary in new settings rather than only how precise the average is. Its main value lies in mapping the evidence across 21 studies (2885 health care workers in 7 countries, covering 5 kinds of AI: note-drafting tools, image-interpretation AI, on-screen decision-support alerts, LLM tools that draft inbox replies, and AI aimed at reducing burnout), and in grading how trustworthy the evidence is for each type of AI. The risk-of-bias assessments and the certainty-of-evidence summary are presented in <xref ref-type="table" rid="table2">Tables 2-4</xref>. Because only 2 to 3 studies could be combined for any single outcome, the pooled numbers are best read as support for this broader map of the evidence, not as conclusions on their own.</p></sec><sec id="s4-2"><title>Meta-Analyzed Outcomes: Ambient AI Documentation</title><p>Our findings should be interpreted within 3 constraints that bear on how much weight they can carry [<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref36">36</xref>]. First, the benefits we observed for ambient AI documentation were not equally firm across outcomes. When the available studies were combined, the reductions in effort, work exhaustion, and burnout were the most consistent. In contrast, the apparent benefits for mental demand and documentation time rested on an insufficient number of studies to be considered established. Furthermore, the studies evaluating mental demand exhibited sufficient variability to preclude treating this reduction as a definitive finding [<xref ref-type="bibr" rid="ref34">34</xref>]. Even for the outcomes that could be examined most fully, the average effect favored ambient AI, though its size in a new clinical setting could range from substantial to small.</p><p>A further and equally important point is that these results speak only to one kind of AI: every pooled outcome came from ambient AI documentation tools. AI as a category-wide remedy for clinician burnout is therefore not supported. The other technologies in this review, including diagnostic imaging AI, CDSS, the LLM inbox tool, and the AI burnout intervention, are discussed narratively and were not pooled. Second, most of the evidence comes from study designs that are more prone to bias. In total, 18 of the 21 studies were non-RCTs, with only 2 RCTs evaluating ambient AI [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. Some of the most pronounced benefits were derived from very small simulation studies [<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref52">52</xref>]. Because participants were largely volunteers and early adopters, the benefits may be overstated. Third, when we formally graded how trustworthy the evidence is using GRADE [<xref ref-type="bibr" rid="ref41">41</xref>], only the reduction in cognitive workload with ambient AI documentation reached moderate confidence. Reductions in burnout and documentation time were graded as low confidence, and the evidence supporting diagnostic imaging AI, CDSS, and the single-trial AI burnout intervention was rated as very low confidence. Whether clinical AI improves the working life of the health care workforce overall (once accuracy, downstream patient safety, and long-term adaptation are taken into account) is not yet established.</p></sec><sec id="s4-3"><title>Comparison With Previous Literature</title><p>Our findings align with and extend the 2024 systematic review [<xref ref-type="bibr" rid="ref15">15</xref>], which identified the absence of cognitive workload assessment as a notable gap in health care AI implementation research. While previous reviews focused primarily on AI diagnostic performance and efficiency metrics [<xref ref-type="bibr" rid="ref15">15</xref>], our review specifically quantified the subjective cognitive experience of clinicians using validated instruments. The observed NASA-TLX reductions of 24&#x2010;40 points with ambient AI scribes exceed the minimally important difference threshold of 10 points suggested in human factors literature, indicating clinically meaningful cognitive burden relief [<xref ref-type="bibr" rid="ref22">22</xref>].</p><p>The unintended workload increase observed with sepsis prediction AI [<xref ref-type="bibr" rid="ref17">17</xref>] (NASA-TLX 43&#x2192;57) corroborates the seminal work of Bainbridge [<xref ref-type="bibr" rid="ref12">12</xref>] on the &#x201C;ironies of automation,&#x201D; predicting that systems designed to reduce human workload often create new cognitive demands through vigilance and oversight requirements. Similarly, the lack of workload reduction despite increased reading time with prostate MRI AI [<xref ref-type="bibr" rid="ref16">16</xref>] supports the &#x201C;out-of-the-loop&#x201D; performance problem [<xref ref-type="bibr" rid="ref13">13</xref>], where monitoring an automated system erodes the operator&#x2019;s situation awareness and capacity to detect the very errors the automation was intended to prevent [<xref ref-type="bibr" rid="ref14">14</xref>].</p><p>The direction of our pooled ambient-documentation findings is concordant with the larger primary literature that did not qualify for meta-analysis. Uncontrolled and pre-post evaluations of commercial ambient scribes have repeatedly reported reduced documentation burden, lower work exhaustion, and improved professional fulfillment [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref44">44</xref>,<xref ref-type="bibr" rid="ref45">45</xref>]. Furthermore, a dedicated cognitive-load evaluation of an ambient platform reported reductions in subjective workload that align with the NASA-TLX effort effect observed in our synthesis [<xref ref-type="bibr" rid="ref43">43</xref>]. These reports, however, share the structural features that constrain our pooled estimate. They are concentrated on a small number of commercial products, conducted in voluntary early-adopter cohorts, and predominantly use uncontrolled pre-post designs carrying a serious risk of bias under ROBINS-I [<xref ref-type="bibr" rid="ref32">32</xref>]. Their convergence should therefore be read as consistent but low-certainty evidence rather than as confirmation of a robust effect, which is the reason the GRADE certainty for burnout reduction with ambient AI documentation was rated low rather than moderate [<xref ref-type="bibr" rid="ref41">41</xref>].</p><p>The 2 RCTs available in this field [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>] provide the least biased evidence; yet, both evaluated a single product over a short horizon, and neither measured downstream documentation accuracy or patient-safety end points. This scarcity of randomized data, combined with the small number of studies available for each outcome [<xref ref-type="bibr" rid="ref34">34</xref>] and the fact that benefit could not be assumed in new settings even where the evidence was strongest [<xref ref-type="bibr" rid="ref35">35</xref>], is the reason our synthesis stops short of asserting a generalizable benefit. The largest workload reductions originated from very small simulation studies [<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref52">52</xref>], whose effect sizes are statistically fragile. This pattern suggests the potential presence of small-study effects [<xref ref-type="bibr" rid="ref40">40</xref>] and reinforces the cautious interpretation demanded by the GRADE assessment.</p></sec><sec id="s4-4"><title>Studies Not Included in the Meta-Analysis: A Narrative Synthesis</title><p>In contrast to the ambient-documentation literature, evidence from diagnostic imaging AI and CDSS is heterogeneous and, in several reports, contrary to expectations. CADe for prostate MRI produced no workload reduction despite improved diagnostic performance [<xref ref-type="bibr" rid="ref16">16</xref>], a pediatric sepsis prediction model increased perceived workload and alert fatigue [<xref ref-type="bibr" rid="ref17">17</xref>], and a mobile CDSS tool improved guideline adherence without lowering mental workload [<xref ref-type="bibr" rid="ref49">49</xref>]. Conversely, a knowledge-recommender embedded in radiology reporting reduced cognitive load [<xref ref-type="bibr" rid="ref47">47</xref>]. This divergence is consistent with the imaging review that first identified the near-absence of workload measurement in this domain [<xref ref-type="bibr" rid="ref15">15</xref>] and with automation-science predictions that alert- and flag-based systems impose monitoring and verification demands distinct from those of generative tools [<xref ref-type="bibr" rid="ref12">12</xref>-<xref ref-type="bibr" rid="ref14">14</xref>]. Because no 2 of these studies measured the same construct with the same instrument in the same setting, they could not be pooled, and the certainty of evidence for imaging AI and CDSS outcomes was correspondingly rated very low [<xref ref-type="bibr" rid="ref41">41</xref>]. This inconsistency is best interpreted as genuine clinical and methodological heterogeneity rather than as evidence of no effect.</p><p>Two further categories of AI could likewise not be pooled and therefore described narratively. An LLM tool that drafts replies to patient inbox messages was evaluated in a single pre-post study, in which perceived task load fell substantially and work exhaustion improved over 5 weeks [<xref ref-type="bibr" rid="ref37">37</xref>]. Because only one study examined this application, the finding is promising but cannot be generalized. It highlights a specific application of AI that differs fundamentally from both clinical note generation and image interpretation. Finally, one RCT examined AI not as a clinical tool but as a means of delivering a personalized burnout intervention. A smartphone app for nurses produced marked reductions in client-related and personal burnout relative to control over 4 weeks [<xref ref-type="bibr" rid="ref51">51</xref>]. This study addresses clinician burnout from a different perspective, using AI to support well-being rather than to assist clinical work. Although its single-trial evidence was graded as low certainty, it illustrates that the role of AI in mitigating burnout extends beyond documentation tools.</p></sec><sec id="s4-5"><title>The Hidden Cost of AI: Rethinking Success Metrics</title><p>The AI categories included in this review differ fundamentally in function and in the clinician&#x2019;s role with the tool. Generative AI tools, such as ambient AI scribes and LLM inbox-message drafts, produce clinician-facing draft text that the user reviews, edits, and approves. Conversely, discriminative AI tools, such as CDSS alerts and CADe diagnostic-imaging flags, push specific alerts that the clinician must verify against ground truth at the point of decision. The AI-based burnout intervention RCT delivers a psychological intervention outside the clinical workflow. We refer to the cognitive cost of reviewing, validating, and reconciling AI-generated outputs with clinical judgment as &#x201C;verification burden.&#x201D; As AI systems become ubiquitous, clinicians transition from content generation to content verification, which may explain the unintended workload increases observed with discriminative-AI tools. This framework, drawn from classical automation theory [<xref ref-type="bibr" rid="ref12">12</xref>-<xref ref-type="bibr" rid="ref14">14</xref>], is offered as hypothesis-generating. Direct measurement of these constructs represents a critical priority for future research.</p><p>A potentially important implication of this review, given the certainty caveats, is that AI implementation success cannot be reduced to diagnostic accuracy or time efficiency alone. Our finding that some AI systems unexpectedly increase cognitive workload despite improving efficiency challenges the prevailing assumption that &#x201C;more AI equals better outcomes.&#x201D; This shift toward supervisory review is a role that human cognitive architecture may be ill-suited to sustain [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref12">12</xref>]. This evidence suggests that regulatory bodies such as the Food and Drug Administration and Conformit&#x00E9; Europ&#x00E9;enne marking authorities should consider incorporating human factors evaluation, including validated cognitive workload assessment, into the AI medical device approval process. Current regulatory frameworks focus predominantly on algorithmic performance metrics, potentially overlooking the real-world cognitive demands imposed on end users. Similarly, health care institutions implementing AI tools should routinely assess cognitive workload and burnout outcomes alongside traditional efficiency metrics [<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref54">54</xref>].</p><p>The rapid adoption of ambient AI documentation reflects an ongoing transformation in clinical practice, as physician AI use doubled from 38% to 66% between 2023 and 2024 [<xref ref-type="bibr" rid="ref1">1</xref>]. Our pooled estimates favor ambient AI documentation in early-adopter outpatient cohorts, but the size of this benefit may vary across settings, and the certainty of the evidence is low to moderate. Therefore, the present evidence supports cautious, evaluation-accompanied adoption rather than category-wide endorsement. Health care systems are increasingly viewing ambient AI scribes as one possible component of broader strategies for the clinician burnout crisis, although high-certainty evidence on the workforce-level impact of such adoption is not yet available [<xref ref-type="bibr" rid="ref2">2</xref>].</p></sec><sec id="s4-6"><title>Strengths</title><p>This review has several strengths. We conducted a comprehensive search across 4 databases and included only studies that measured cognitive workload or burnout with validated instruments. We applied a rigorous risk of bias assessment using domain-appropriate tools, specifically RoB 2.0 for RCTs and ROBINS-I for non-RCTs, as detailed in <xref ref-type="table" rid="table2">Tables 2</xref> and <xref ref-type="table" rid="table3">3</xref>. Furthermore, we used a deliberately cautious approach to combining results across all outcomes to avoid overstating uncertain findings. We provided a clear grading of evidence certainty for each outcome, presented in <xref ref-type="table" rid="table4">Table 4</xref>, and included forest plots for every pool in <xref ref-type="fig" rid="figure3">Figures 3</xref><xref ref-type="fig" rid="figure4"/><xref ref-type="fig" rid="figure5"/><xref ref-type="fig" rid="figure6"/><xref ref-type="fig" rid="figure7"/>-<xref ref-type="fig" rid="figure8">8</xref>. Additional checks confirmed that our main conclusions did not depend on the specific analytical assumptions applied in the synthesis.</p></sec><sec id="s4-7"><title>Limitations</title><p>Several caveats apply to the interpretation of our findings. The positive evidence for ambient AI documentation rests on a narrow base. It is concentrated on 2 commercial products (Abridge and DAX Copilot), evaluated predominantly among volunteers and early adopters in US outpatient primary care settings, and followed for a short time (typically 8&#x2010;12 weeks). Because each outcome could be based on only a few studies, these pooled results are best read as preliminary rather than definitive, and the benefits for 2 of the 6 outcomes are no longer clear-cut once this limited evidence is taken into account. For this reason, we judged the certainty of the evidence for burnout reduction to be low rather than moderate, given the reliance on observational studies and a small literature dominated by 2 products with uniformly positive results. Throughout, claims are framed cautiously, with ambient AI documentation described as &#x201C;associated with reductions in&#x201D; cognitive workload and burnout rather than as reliably reducing them.</p><p>Finally, none of the 21 studies directly measured the subtler costs of working with AI. These unmeasured costs include the effort required to verify its output, the tendency to overtrust the system, and the subsequent loss of vigilance. As a result, the original question behind this review (whether AI genuinely reduces the mental burden of clinicians or simply shifts the demand from documentation to verification) cannot be answered with the present evidence and remains a priority for future research.</p><p>The following specific limitations should also be acknowledged. First, each outcome could be based on only a small number of studies, which limits how precisely the effects can be estimated, prevents the planned subgroup and related exploratory analyses [<xref ref-type="bibr" rid="ref55">55</xref>], and leaves considerable uncertainty. Even for the 2 outcomes that could be examined most fully, work exhaustion and burnout prevalence, the average effect favored ambient AI, although the magnitude of this effect may differ in a new clinical setting. Second, the studies of some outcomes, most notably mental demand, varied considerably among themselves. This variability likely occurred because the studies used different measurement scales and were conducted across distinct clinical environments. Furthermore, combining results from abbreviated or modified versions of the same questionnaire adds further inconsistency [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref57">57</xref>]. Third, no study used the full MBI, even though it is the most thoroughly validated measure of burnout [<xref ref-type="bibr" rid="ref25">25</xref>]. Fourth, we cannot rule out that smaller studies reported more favorable results than larger ones. Among the possible explanations, publication bias remains plausible because negative or null studies may be underrepresented in this rapidly commercializing field [<xref ref-type="bibr" rid="ref40">40</xref>].</p></sec><sec id="s4-8"><title>Future Research Directions</title><p>The current evidence base, while suggestive of meaningful benefits from ambient AI documentation, remains preliminary, product-concentrated, and predominantly composed of short-term, single-institution evaluations conducted in early-adopter settings. Although ambient AI scribes appear to reduce documentation time and burnout reports, the present evidence does not allow us to verify documentation accuracy, downstream patient-safety end points, or net benefit once verification burden, automation bias, and long-term cognitive adaptation are accounted for.</p><p>Several research priorities must be addressed before strong clinical or policy recommendations can be formulated. First, future studies should use RCTs with preregistered protocols that pair clinician-reported workload and burnout outcomes with downstream patient-safety, documentation-accuracy, and diagnostic-error end points. Currently, only 4 of the 21 included studies used registered protocols. Second, prospective cohorts extending beyond 12 months are necessary to detect deskilling, adaptation, and longer-term cognitive effects. Third, researchers must prioritize the standardized measurement of verification burden, automation bias, automation complacency, and trust calibration using validated instruments. While narrower constructs such as trust, usability, and alert fatigue were assessed in some studies, the broader cognitive-cost framework remains unmeasured. Fourth, head-to-head comparative effectiveness studies across multiple commercial ambient AI scribe products, beyond early market leaders, are needed. Fifth, investigations should expand to non-US health care systems and specialties outside outpatient primary care, where workflow requirements and clinician burnout drivers may differ. Finally, future analyses should decompose the NASA-TLX into its specific subscales rather than relying on total scores, allowing for the precise identification of which cognitive domains AI affects most. Without such evidence, the workforce-level benefit of ambient AI scribes cannot yet be confirmed.</p></sec><sec id="s4-9"><title>Conclusions</title><p>This systematic review and meta-analysis is, to our knowledge, the first to combine validated cognitive-workload instruments with uniformly adjusted random-effects meta-analysis and PIs in human-AI interaction in health care. It is also the first to grade the certainty of evidence separately for 5 distinct AI categories: ambient documentation, diagnostic imaging AI, CDSS, LLM inbox tools, and AI-based burnout interventions. Whereas prior systematic reviews of health care AI have focused on diagnostic accuracy, efficiency, or implementation outcomes, the present synthesis quantifies the subjective cognitive experience and burnout of clinicians using validated instruments and grades the certainty of every quantitative outcome.</p><p>The contribution to the field is therefore a structured certainty-graded evidence base for the cognitive and burnout consequences of clinical AI. Ambient AI documentation is associated with reductions in NASA-TLX effort, PFI work exhaustion, and burnout prevalence in early-adopter cohorts; however, the evidence is still limited and uncertain. The observed benefits rest on a small number of studies, the effect sizes may not be uniform across settings, and technologies such as diagnostic imaging AI and CDSS have demonstrated context-dependent or unintended workload increases.</p><p>The real-world implications are immediate and concrete. Institutional pilots of ambient AI should be paired with prospective, validated-instrument measurement of cognitive workload and burnout. Regulatory human-factors review must be integrated alongside algorithmic-performance evaluation during AI medical-device approval. Furthermore, AI tool developers should incorporate human-centered design that anticipates verification burden as a primary outcome, a metric that was not directly measured in any included study. Until multiproduct, longer-term, multiregion evidence directly measures verification burden, automation bias, and downstream patient-safety end points, the net benefit of clinical AI on the health care workforce remains an open empirical question.</p></sec></sec></body><back><ack><p>The authors declare the use of generative artificial intelligence (GAI) in the research and writing process. According to the GAIDeT taxonomy (2025), the following tasks were delegated to GAI tools under full human supervision: proofreading and editing, adapting and adjusting emotional tone, translation, and reformatting. The GAI tool used was Claude 4.8. Responsibility for the final manuscript lies entirely with the authors. GAI tools are not listed as authors and do not bear responsibility for the final outcomes.</p></ack><notes><sec><title>Funding</title><p>This research was supported by the Bio&#x0026;Medical Technology Development Program of the National Research Foundation funded by the Korean government (MSIT; RS2023-00223501).</p></sec><sec><title>Data Availability</title><p>All data generated or analyzed during this study are included in this paper.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: CSB</p><p>Data curation: CSB, EJG, JJL</p><p>Formal analysis: CSB</p><p>Funding acquisition: JJL</p><p>Investigation: CSB, EJG, JJL</p><p>Methodology: CSB</p><p>Project administration: CSB</p><p>Resources: CSB</p><p>Writing&#x2014;original draft: EJG, CSB</p><p>Writing&#x2014;review and editing: CSB, JJL</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AI</term><def><p>artificial intelligence</p></def></def-item><def-item><term id="abb2">APP</term><def><p>advanced practice provider</p></def></def-item><def-item><term id="abb3">CADe</term><def><p>computer-aided detection</p></def></def-item><def-item><term id="abb4">CBI</term><def><p>Copenhagen Burnout Inventory</p></def></def-item><def-item><term id="abb5">CDSS</term><def><p>clinical decision support system</p></def></def-item><def-item><term id="abb6">GRADE</term><def><p>Grading of Recommendations Assessment, Development and Evaluation</p></def></def-item><def-item><term id="abb7">HKSJ</term><def><p>Hartung-Knapp-Sidik-Jonkman</p></def></def-item><def-item><term id="abb8">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb9">MBI</term><def><p>Maslach Burnout Inventory</p></def></def-item><def-item><term id="abb10">MD</term><def><p>mean difference</p></def></def-item><def-item><term id="abb11">MRI</term><def><p>magnetic resonance imaging</p></def></def-item><def-item><term id="abb12">NASA-TLX</term><def><p>National Aeronautics and Space Administration Task Load Index</p></def></def-item><def-item><term id="abb13">OLBI</term><def><p>Oldenburg Burnout Inventory</p></def></def-item><def-item><term id="abb14">OR</term><def><p>odds ratio</p></def></def-item><def-item><term id="abb15">PFI</term><def><p>Professional Fulfillment Index</p></def></def-item><def-item><term id="abb16">PI</term><def><p>prediction interval</p></def></def-item><def-item><term id="abb17">PRISMA</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses</p></def></def-item><def-item><term id="abb18">PRISMA-S</term><def><p>Preferred Reporting Items for Systematic reviews and Meta-Analyses Literature Search Extension</p></def></def-item><def-item><term id="abb19">RCT</term><def><p>randomized controlled trial</p></def></def-item><def-item><term id="abb20">RoB 2</term><def><p>Risk of Bias tool version 2.0</p></def></def-item><def-item><term id="abb21">ROBINS-I</term><def><p>Risk of Bias in Non-randomized Studies of Interventions</p></def></def-item><def-item><term id="abb22">SMD</term><def><p>standardized mean difference</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="web"><article-title>2026 Physician survey on augmented intelligence</article-title><source>American Medical Association</source><year>2025</year><month>02</month><access-date>2026-07-16</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.ama-assn.org/system/files/physician-ai-sentiment-report.pdf">https://www.ama-assn.org/system/files/physician-ai-sentiment-report.pdf</ext-link></comment></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shanafelt</surname><given-names>TD</given-names> </name><name name-style="western"><surname>West</surname><given-names>CP</given-names> </name><name name-style="western"><surname>Sinsky</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Changes in burnout and satisfaction with work-life integration in physicians and the general US working population between 2011 and 2023</article-title><source>Mayo Clin Proc</source><year>2025</year><month>07</month><volume>100</volume><issue>7</issue><fpage>1142</fpage><lpage>1158</lpage><pub-id pub-id-type="doi">10.1016/j.mayocp.2024.11.031</pub-id><pub-id pub-id-type="medline">40202475</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sinsky</surname><given-names>C</given-names> </name><name name-style="western"><surname>Colligan</surname><given-names>L</given-names> </name><name name-style="western"><surname>Li</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Allocation of physician time in ambulatory practice: a time and motion study in 4 specialties</article-title><source>Ann Intern Med</source><year>2016</year><month>12</month><day>6</day><volume>165</volume><issue>11</issue><fpage>753</fpage><lpage>760</lpage><pub-id pub-id-type="doi">10.7326/M16-0961</pub-id><pub-id pub-id-type="medline">27595430</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Olson</surname><given-names>KD</given-names> </name><name name-style="western"><surname>Meeker</surname><given-names>D</given-names> </name><name name-style="western"><surname>Troup</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Use of ambient AI scribes to reduce administrative burden and professional burnout</article-title><source>JAMA Netw Open</source><year>2025</year><month>10</month><day>1</day><volume>8</volume><issue>10</issue><fpage>e2534976</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2025.34976</pub-id><pub-id pub-id-type="medline">41037268</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tierney</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Gayre</surname><given-names>G</given-names> </name><name name-style="western"><surname>Hoberman</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Ambient artificial intelligence scribes to alleviate the burden of clinical documentation</article-title><source>NEJM Catalyst</source><year>2024</year><month>02</month><day>21</day><volume>5</volume><issue>3</issue><fpage>0404</fpage><pub-id pub-id-type="doi">10.1056/CAT.23.0404</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Albrecht</surname><given-names>M</given-names> </name><name name-style="western"><surname>Shanks</surname><given-names>D</given-names> </name><name name-style="western"><surname>Shah</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Enhancing clinical documentation with ambient artificial intelligence: a quality improvement survey assessing clinician perspectives on work burden, burnout, and job satisfaction</article-title><source>JAMIA Open</source><year>2025</year><month>02</month><volume>8</volume><issue>1</issue><fpage>ooaf013</fpage><pub-id pub-id-type="doi">10.1093/jamiaopen/ooaf013</pub-id><pub-id pub-id-type="medline">39991073</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lukac</surname><given-names>PJ</given-names> </name><name name-style="western"><surname>Turner</surname><given-names>W</given-names> </name><name name-style="western"><surname>Vangala</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Ambient AI scribes in clinical practice: a randomized trial</article-title><source>NEJM AI</source><year>2025</year><month>12</month><volume>2</volume><issue>12</issue><pub-id pub-id-type="doi">10.1056/aioa2501000</pub-id><pub-id pub-id-type="medline">41497288</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Afshar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Baumann</surname><given-names>MR</given-names> </name><name name-style="western"><surname>Resnik</surname><given-names>F</given-names> </name><etal/></person-group><article-title>A pragmatic randomized controlled trial of ambient artificial intelligence to improve health practitioner well-being</article-title><source>NEJM AI</source><year>2025</year><month>12</month><volume>2</volume><issue>12</issue><fpage>10</fpage><pub-id pub-id-type="doi">10.1056/aioa2500945</pub-id><pub-id pub-id-type="medline">41625485</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>You</surname><given-names>JG</given-names> </name><name name-style="western"><surname>Dbouk</surname><given-names>RH</given-names> </name><name name-style="western"><surname>Landman</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Ambient documentation technology in clinician experience of documentation burden and burnout</article-title><source>JAMA Netw Open</source><year>2025</year><month>08</month><day>1</day><volume>8</volume><issue>8</issue><fpage>e2528056</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2025.28056</pub-id><pub-id pub-id-type="medline">40839265</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Stults</surname><given-names>CD</given-names> </name><name name-style="western"><surname>Deng</surname><given-names>S</given-names> </name><name name-style="western"><surname>Martinez</surname><given-names>MC</given-names> </name><etal/></person-group><article-title>Evaluation of an ambient artificial intelligence documentation platform for clinicians</article-title><source>JAMA Netw Open</source><year>2025</year><month>05</month><day>1</day><volume>8</volume><issue>5</issue><fpage>e258614</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2025.8614</pub-id><pub-id pub-id-type="medline">40314951</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Oppenheimer</surname><given-names>DM</given-names> </name></person-group><article-title>The secret life of fluency</article-title><source>Trends Cogn Sci</source><year>2008</year><month>06</month><volume>12</volume><issue>6</issue><fpage>237</fpage><lpage>241</lpage><pub-id pub-id-type="doi">10.1016/j.tics.2008.02.014</pub-id><pub-id pub-id-type="medline">18468944</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Bainbridge</surname><given-names>L</given-names> </name></person-group><source>Analysis, Design and Evaluation of Man&#x2013;Machine Systems</source><year>1983</year><publisher-name>Elsevier</publisher-name><fpage>129</fpage><lpage>135</lpage><pub-id pub-id-type="doi">10.1016/B978-0-08-029348-6.50026-9</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Endsley</surname><given-names>MR</given-names> </name><name name-style="western"><surname>Kiris</surname><given-names>EO</given-names> </name></person-group><article-title>The out-of-the-loop performance problem and level of control in automation</article-title><source>Hum Factors</source><year>1995</year><month>06</month><volume>37</volume><issue>2</issue><fpage>381</fpage><lpage>394</lpage><pub-id pub-id-type="doi">10.1518/001872095779064555</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Parasuraman</surname><given-names>R</given-names> </name><name name-style="western"><surname>Manzey</surname><given-names>DH</given-names> </name></person-group><article-title>Complacency and bias in human use of automation: an attentional integration</article-title><source>Hum Factors</source><year>2010</year><month>06</month><volume>52</volume><issue>3</issue><fpage>381</fpage><lpage>410</lpage><pub-id pub-id-type="doi">10.1177/0018720810376055</pub-id><pub-id pub-id-type="medline">21077562</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wenderott</surname><given-names>K</given-names> </name><name name-style="western"><surname>Krups</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zaruchas</surname><given-names>F</given-names> </name><name name-style="western"><surname>Weigl</surname><given-names>M</given-names> </name></person-group><article-title>Effects of artificial intelligence implementation on efficiency in medical imaging-a systematic literature review and meta-analysis</article-title><source>NPJ Digit Med</source><year>2024</year><month>09</month><day>30</day><volume>7</volume><issue>1</issue><fpage>265</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01248-9</pub-id><pub-id pub-id-type="medline">39349815</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wenderott</surname><given-names>K</given-names> </name><name name-style="western"><surname>Krups</surname><given-names>J</given-names> </name><name name-style="western"><surname>Luetkens</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Gambashidze</surname><given-names>N</given-names> </name><name name-style="western"><surname>Weigl</surname><given-names>M</given-names> </name></person-group><article-title>Prospective effects of an artificial intelligence-based computer-aided detection system for prostate imaging on routine workflow and radiologists&#x2019; outcomes</article-title><source>Eur J Radiol</source><year>2024</year><month>01</month><volume>170</volume><fpage>111252</fpage><pub-id pub-id-type="doi">10.1016/j.ejrad.2023.111252</pub-id><pub-id pub-id-type="medline">38096741</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kandaswamy</surname><given-names>S</given-names> </name><name name-style="western"><surname>Muthu</surname><given-names>N</given-names> </name><name name-style="western"><surname>Braykov</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Human performance evaluation of a pediatric artificial intelligence sepsis model</article-title><source>J Am Med Inform Assoc</source><year>2025</year><month>10</month><day>1</day><volume>32</volume><issue>10</issue><fpage>1552</fpage><lpage>1561</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocaf106</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>West</surname><given-names>CP</given-names> </name><name name-style="western"><surname>Dyrbye</surname><given-names>LN</given-names> </name><name name-style="western"><surname>Shanafelt</surname><given-names>TD</given-names> </name></person-group><article-title>Physician burnout: contributors, consequences and solutions</article-title><source>J Intern Med</source><year>2018</year><month>06</month><volume>283</volume><issue>6</issue><fpage>516</fpage><lpage>529</lpage><pub-id pub-id-type="doi">10.1111/joim.12752</pub-id><pub-id pub-id-type="medline">29505159</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Page</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>McKenzie</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Bossuyt</surname><given-names>PM</given-names> </name><etal/></person-group><article-title>The PRISMA 2020 statement: an updated guideline for reporting systematic reviews</article-title><source>BMJ</source><year>2021</year><volume>372</volume><fpage>n71</fpage><pub-id pub-id-type="doi">10.1136/bmj.n71</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rethlefsen</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Kirtley</surname><given-names>S</given-names> </name><name name-style="western"><surname>Waffenschmidt</surname><given-names>S</given-names> </name><etal/></person-group><article-title>PRISMA-S: an extension to the PRISMA Statement for Reporting Literature Searches in Systematic Reviews</article-title><source>Syst Rev</source><year>2021</year><month>01</month><day>26</day><volume>10</volume><issue>1</issue><fpage>39</fpage><pub-id pub-id-type="doi">10.1186/s13643-020-01542-z</pub-id><pub-id pub-id-type="medline">33499930</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Eriksen</surname><given-names>MB</given-names> </name><name name-style="western"><surname>Frandsen</surname><given-names>TF</given-names> </name></person-group><article-title>The impact of patient, intervention, comparison, outcome (PICO) as a search strategy tool on literature search quality: a systematic review</article-title><source>J Med Libr Assoc</source><year>2018</year><month>10</month><volume>106</volume><issue>4</issue><fpage>420</fpage><lpage>431</lpage><pub-id pub-id-type="doi">10.5195/jmla.2018.345</pub-id><pub-id pub-id-type="medline">30271283</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hart</surname><given-names>SG</given-names> </name><name name-style="western"><surname>Staveland</surname><given-names>LE</given-names> </name></person-group><article-title>Development of NASA-TLX (Task Load Index): results of empirical and theoretical research</article-title><source>Adv Psychol</source><year>1988</year><volume>52</volume><fpage>139</fpage><lpage>183</lpage><pub-id pub-id-type="doi">10.1016/S0166-4115(08)62386-9</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Reid</surname><given-names>GB</given-names> </name><name name-style="western"><surname>Nygren</surname><given-names>TE</given-names> </name></person-group><article-title>The subjective workload assessment technique: a scaling procedure for measuring mental workload</article-title><source>Adv Psychol</source><year>1988</year><volume>52</volume><fpage>185</fpage><lpage>218</lpage><pub-id pub-id-type="doi">10.1016/S0166-4115(08)62387-0</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Paas</surname><given-names>F</given-names> </name></person-group><article-title>Training strategies for attaining transfer of problem-solving skill in statistics: a cognitive-load approach</article-title><source>J Educ Psychol</source><year>1992</year><volume>84</volume><issue>4</issue><fpage>429</fpage><lpage>434</lpage><pub-id pub-id-type="doi">10.1037/0022-0663.84.4.429</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Maslach</surname><given-names>C</given-names> </name><name name-style="western"><surname>Jackson</surname><given-names>SE</given-names> </name></person-group><article-title>The measurement of experienced burnout</article-title><source>J Organ Behavior</source><year>1981</year><month>04</month><volume>2</volume><issue>2</issue><fpage>99</fpage><lpage>113</lpage><pub-id pub-id-type="doi">10.1002/job.4030020205</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Demerouti</surname><given-names>E</given-names> </name><name name-style="western"><surname>Bakker</surname><given-names>AB</given-names> </name><name name-style="western"><surname>Nachreiner</surname><given-names>F</given-names> </name><name name-style="western"><surname>Schaufeli</surname><given-names>WB</given-names> </name></person-group><article-title>The job demands-resources model of burnout</article-title><source>J Appl Psychol</source><year>2001</year><month>06</month><volume>86</volume><issue>3</issue><fpage>499</fpage><lpage>512</lpage><pub-id pub-id-type="medline">11419809</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Trockel</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bohman</surname><given-names>B</given-names> </name><name name-style="western"><surname>Lesure</surname><given-names>E</given-names> </name><etal/></person-group><article-title>A brief instrument to assess both burnout and professional fulfillment in physicians: reliability and validity, including correlation with self-reported medical errors, in a sample of resident and practicing physicians</article-title><source>Acad Psychiatry</source><year>2018</year><month>02</month><volume>42</volume><issue>1</issue><fpage>11</fpage><lpage>24</lpage><pub-id pub-id-type="doi">10.1007/s40596-017-0849-3</pub-id><pub-id pub-id-type="medline">29196982</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kristensen</surname><given-names>TS</given-names> </name><name name-style="western"><surname>Borritz</surname><given-names>M</given-names> </name><name name-style="western"><surname>Villadsen</surname><given-names>E</given-names> </name><name name-style="western"><surname>Christensen</surname><given-names>KB</given-names> </name></person-group><article-title>The Copenhagen Burnout Inventory: a new tool for the assessment of burnout</article-title><source>Work Stress</source><year>2005</year><month>07</month><volume>19</volume><issue>3</issue><fpage>192</fpage><lpage>207</lpage><pub-id pub-id-type="doi">10.1080/02678370500297720</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Linzer</surname><given-names>M</given-names> </name><name name-style="western"><surname>Poplau</surname><given-names>S</given-names> </name><name name-style="western"><surname>Babbott</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Worklife and wellness in academic general internal medicine: results from a national survey</article-title><source>J Gen Intern Med</source><year>2016</year><month>09</month><volume>31</volume><issue>9</issue><fpage>1004</fpage><lpage>1010</lpage><pub-id pub-id-type="doi">10.1007/s11606-016-3720-4</pub-id><pub-id pub-id-type="medline">27138425</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Brooke</surname><given-names>J</given-names> </name></person-group><article-title>SUS-a quick and dirty usability scale</article-title><source>Usability Evaluation in Industry</source><year>1996</year><volume>189</volume><publisher-name>CRC Press</publisher-name><fpage>4</fpage><lpage>7</lpage><pub-id pub-id-type="doi">10.1201/9781498710411</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sterne</surname><given-names>JAC</given-names> </name><name name-style="western"><surname>Savovi&#x0107;</surname><given-names>J</given-names> </name><name name-style="western"><surname>Page</surname><given-names>MJ</given-names> </name><etal/></person-group><article-title>RoB 2: a revised tool for assessing risk of bias in randomised trials</article-title><source>BMJ</source><year>2019</year><month>08</month><day>28</day><volume>366</volume><fpage>l4898</fpage><pub-id pub-id-type="doi">10.1136/bmj.l4898</pub-id><pub-id pub-id-type="medline">31462531</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sterne</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Hern&#x00E1;n</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Reeves</surname><given-names>BC</given-names> </name><etal/></person-group><article-title>ROBINS-I: a tool for assessing risk of bias in non-randomised studies of interventions</article-title><source>BMJ</source><year>2016</year><month>10</month><day>12</day><volume>355</volume><fpage>i4919</fpage><pub-id pub-id-type="doi">10.1136/bmj.i4919</pub-id><pub-id pub-id-type="medline">27733354</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Popay</surname><given-names>J</given-names> </name><name name-style="western"><surname>Roberts</surname><given-names>H</given-names> </name><name name-style="western"><surname>Sowden</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Guidance on the conduct of narrative synthesis in systematic reviews. A product from the ESRC methods programme version 2006</article-title><access-date>2026-07-16</access-date><publisher-name>Lancaster University</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.york.ac.uk/media/crd/Guidance%20on%20the%20conduct%20of%20narrative%20synthesis%20in%20systematic%20review.pdf">https://www.york.ac.uk/media/crd/Guidance%20on%20the%20conduct%20of%20narrative%20synthesis%20in%20systematic%20review.pdf</ext-link></comment></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>IntHout</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ioannidis</surname><given-names>JPA</given-names> </name><name name-style="western"><surname>Borm</surname><given-names>GF</given-names> </name></person-group><article-title>The Hartung-Knapp-Sidik-Jonkman method for random effects meta-analysis is straightforward and considerably outperforms the standard DerSimonian-Laird method</article-title><source>BMC Med Res Methodol</source><year>2014</year><month>02</month><day>18</day><volume>14</volume><fpage>25</fpage><pub-id pub-id-type="doi">10.1186/1471-2288-14-25</pub-id><pub-id pub-id-type="medline">24548571</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Borenstein</surname><given-names>M</given-names> </name></person-group><article-title>How to understand and report heterogeneity in a meta-analysis: the difference between I-squared and prediction intervals</article-title><source>Integr Med Res</source><year>2023</year><month>12</month><volume>12</volume><issue>4</issue><fpage>101014</fpage><pub-id pub-id-type="doi">10.1016/j.imr.2023.101014</pub-id><pub-id pub-id-type="medline">38938910</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nagashima</surname><given-names>K</given-names> </name><name name-style="western"><surname>Noma</surname><given-names>H</given-names> </name><name name-style="western"><surname>Furukawa</surname><given-names>TA</given-names> </name></person-group><article-title>Prediction intervals for random-effects meta-analysis: a confidence distribution approach</article-title><source>Stat Methods Med Res</source><year>2019</year><month>06</month><volume>28</volume><issue>6</issue><fpage>1689</fpage><lpage>1702</lpage><pub-id pub-id-type="doi">10.1177/0962280218773520</pub-id><pub-id pub-id-type="medline">29745296</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Garcia</surname><given-names>P</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>SP</given-names> </name><name name-style="western"><surname>Shah</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Artificial intelligence-generated draft replies to patient inbox messages</article-title><source>JAMA Netw Open</source><year>2024</year><month>03</month><day>4</day><volume>7</volume><issue>3</issue><fpage>e243201</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.3201</pub-id><pub-id pub-id-type="medline">38506805</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pelletier</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Watson</surname><given-names>K</given-names> </name><name name-style="western"><surname>Michel</surname><given-names>J</given-names> </name><name name-style="western"><surname>McGregor</surname><given-names>R</given-names> </name><name name-style="western"><surname>Rush</surname><given-names>SZ</given-names> </name></person-group><article-title>Effect of a generative artificial intelligence digital scribe on pediatric provider documentation time, cognitive burden, and burnout</article-title><source>JAMIA Open</source><year>2025</year><month>08</month><volume>8</volume><issue>4</issue><fpage>ooaf068</fpage><pub-id pub-id-type="doi">10.1093/jamiaopen/ooaf068</pub-id><pub-id pub-id-type="medline">40620477</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Duggan</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Gervase</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schoenbaum</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Clinician experiences with ambient scribe technology to assist with documentation burden and efficiency</article-title><source>JAMA Netw Open</source><year>2025</year><month>02</month><day>3</day><volume>8</volume><issue>2</issue><fpage>e2460637</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.60637</pub-id><pub-id pub-id-type="medline">39969880</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sterne</surname><given-names>JAC</given-names> </name><name name-style="western"><surname>Sutton</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Ioannidis</surname><given-names>JPA</given-names> </name><etal/></person-group><article-title>Recommendations for examining and interpreting funnel plot asymmetry in meta-analyses of randomised controlled trials</article-title><source>BMJ</source><year>2011</year><month>07</month><day>22</day><volume>343</volume><fpage>d4002</fpage><pub-id pub-id-type="doi">10.1136/bmj.d4002</pub-id><pub-id pub-id-type="medline">21784880</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guyatt</surname><given-names>GH</given-names> </name><name name-style="western"><surname>Oxman</surname><given-names>AD</given-names> </name><name name-style="western"><surname>Vist</surname><given-names>GE</given-names> </name><etal/></person-group><article-title>GRADE: an emerging consensus on rating quality of evidence and strength of recommendations</article-title><source>BMJ</source><year>2008</year><month>04</month><day>26</day><volume>336</volume><issue>7650</issue><fpage>924</fpage><lpage>926</lpage><pub-id pub-id-type="doi">10.1136/bmj.39489.470347.AD</pub-id><pub-id pub-id-type="medline">18436948</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shah</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Devon-Sand</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>SP</given-names> </name><etal/></person-group><article-title>Ambient artificial intelligence scribes: physician burnout and perspectives on usability and documentation burden</article-title><source>J Am Med Inform Assoc</source><year>2025</year><month>02</month><day>1</day><volume>32</volume><issue>2</issue><fpage>375</fpage><lpage>380</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocae295</pub-id><pub-id pub-id-type="medline">39657021</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hudson</surname><given-names>TJ</given-names> </name><name name-style="western"><surname>Albrecht</surname><given-names>M</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>TR</given-names> </name><etal/></person-group><article-title>Impact of ambient artificial intelligence documentation on cognitive load</article-title><source>Mayo Clin Proc Digit Health</source><year>2025</year><month>03</month><volume>3</volume><issue>1</issue><fpage>100193</fpage><pub-id pub-id-type="doi">10.1016/j.mcpdig.2024.100193</pub-id><pub-id pub-id-type="medline">40206994</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Owens</surname><given-names>LM</given-names> </name><name name-style="western"><surname>Wilda</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Hahn</surname><given-names>PY</given-names> </name><name name-style="western"><surname>Koehler</surname><given-names>T</given-names> </name><name name-style="western"><surname>Fletcher</surname><given-names>JJ</given-names> </name></person-group><article-title>The association between use of ambient voice technology documentation during primary care patient encounters, documentation burden, and provider burnout</article-title><source>Fam Pract</source><year>2024</year><month>04</month><day>15</day><volume>41</volume><issue>2</issue><fpage>86</fpage><lpage>91</lpage><pub-id pub-id-type="doi">10.1093/fampra/cmad092</pub-id><pub-id pub-id-type="medline">37672297</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Misurac</surname><given-names>J</given-names> </name><name name-style="western"><surname>Knake</surname><given-names>LA</given-names> </name><name name-style="western"><surname>Blum</surname><given-names>JM</given-names> </name></person-group><article-title>The effect of ambient artificial intelligence notes on provider burnout</article-title><source>Appl Clin Inform</source><year>2025</year><month>03</month><volume>16</volume><issue>2</issue><fpage>252</fpage><lpage>258</lpage><pub-id pub-id-type="doi">10.1055/a-2461-4576</pub-id><pub-id pub-id-type="medline">39500346</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bracken</surname><given-names>A</given-names> </name><name name-style="western"><surname>Babu</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Whelehan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Merghani</surname><given-names>K</given-names> </name><name name-style="western"><surname>Sheehan</surname><given-names>E</given-names> </name><name name-style="western"><surname>Feeley</surname><given-names>I</given-names> </name></person-group><article-title>Ambient AI reduces documentation time and enhances quality in a simulated inpatient setting</article-title><source>Surgeon</source><year>2026</year><month>04</month><volume>24</volume><issue>2</issue><fpage>119</fpage><lpage>125</lpage><pub-id pub-id-type="doi">10.1016/j.surge.2025.10.008</pub-id><pub-id pub-id-type="medline">41198484</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lopez-Rippe</surname><given-names>J</given-names> </name><name name-style="western"><surname>Reddy</surname><given-names>M</given-names> </name><name name-style="western"><surname>Velez-Florez</surname><given-names>MC</given-names> </name><etal/></person-group><article-title>RADHawk&#x2014;an AI-based knowledge recommender to support precision education, improve reporting productivity, and reduce cognitive load</article-title><source>Pediatr Radiol</source><year>2025</year><volume>55</volume><issue>2</issue><fpage>259</fpage><lpage>267</lpage><pub-id pub-id-type="doi">10.1007/s00247-024-06116-y</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Muzumala</surname><given-names>MG</given-names> </name><name name-style="western"><surname>Zulu</surname><given-names>EO</given-names> </name><name name-style="western"><surname>Chibuta</surname><given-names>P</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Wu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Shabestari</surname><given-names>B</given-names> </name><name name-style="western"><surname>Xing</surname><given-names>L</given-names> </name></person-group><article-title>Evaluating perceived workload, usability and usefulness of artificial intelligence systems in low-resource settings: semi-automated classification and detection of community acquired pneumonia</article-title><source>Applications of Medical Artificial Intelligence AMAI 2024 Lecture Notes in Computer Science</source><publisher-name>Springer</publisher-name><pub-id pub-id-type="doi">10.1007/978-3-031-82007-6_12</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Richardson</surname><given-names>KM</given-names> </name><name name-style="western"><surname>Fouquet</surname><given-names>SD</given-names> </name><name name-style="western"><surname>Kerns</surname><given-names>E</given-names> </name><name name-style="western"><surname>McCulloh</surname><given-names>RJ</given-names> </name></person-group><article-title>Impact of mobile device-based clinical decision support tool on guideline adherence and mental workload</article-title><source>Acad Pediatr</source><year>2019</year><volume>19</volume><issue>7</issue><fpage>828</fpage><lpage>834</lpage><pub-id pub-id-type="doi">10.1016/j.acap.2019.03.001</pub-id><pub-id pub-id-type="medline">30853573</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sanderson</surname><given-names>BJ</given-names> </name><name name-style="western"><surname>Field</surname><given-names>JD</given-names> </name><name name-style="western"><surname>Kocaballi</surname><given-names>AB</given-names> </name><etal/></person-group><article-title>Clinical decision support versus a paper-based protocol for massive transfusion: impact on decision outcomes in a simulation study</article-title><source>Transfusion</source><year>2023</year><month>12</month><volume>63</volume><issue>12</issue><fpage>2225</fpage><lpage>2233</lpage><pub-id pub-id-type="doi">10.1111/trf.17580</pub-id><pub-id pub-id-type="medline">37921017</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Baek</surname><given-names>G</given-names> </name><name name-style="western"><surname>Cha</surname><given-names>C</given-names> </name></person-group><article-title>AI-assisted tailored intervention for nurse burnout: a three-group randomized controlled trial</article-title><source>Worldviews Evid Based Nurs</source><year>2025</year><month>02</month><volume>22</volume><issue>1</issue><fpage>e70003</fpage><pub-id pub-id-type="doi">10.1111/wvn.70003</pub-id><pub-id pub-id-type="medline">39981583</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Nawaz</surname><given-names>FA</given-names> </name><name name-style="western"><surname>Bokhari</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Usman</surname><given-names>FM</given-names> </name><etal/></person-group><article-title>Evaluating an ambient artificial intelligence scribe for documentation quality and efficiency in psychiatric consultations: a simulation-based study</article-title><source>medRxiv</source><comment>Preprint posted online on  Sep 22, 2025</comment><pub-id pub-id-type="doi">10.1101/2025.09.21.25336260</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gong</surname><given-names>EJ</given-names> </name><name name-style="western"><surname>Bang</surname><given-names>CS</given-names> </name></person-group><article-title>Clinical implementation of artificial intelligence in endoscopy: a human-artificial intelligence interaction perspective</article-title><source>Korean J Gastroenterol</source><year>2026</year><month>01</month><day>25</day><volume>86</volume><issue>1</issue><fpage>1</fpage><lpage>9</lpage><pub-id pub-id-type="doi">10.4166/kjg.2025.151</pub-id><pub-id pub-id-type="medline">41572653</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gong</surname><given-names>EJ</given-names> </name><name name-style="western"><surname>Bang</surname><given-names>CS</given-names> </name></person-group><article-title>Artificial intelligence in colonoscopy: polyp fiction or clinical reality?</article-title><source>Clin Endosc</source><year>2025</year><month>09</month><volume>58</volume><issue>5</issue><fpage>784</fpage><lpage>786</lpage><pub-id pub-id-type="doi">10.5946/ce.2025.103</pub-id><pub-id pub-id-type="medline">40899245</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Higgins</surname><given-names>JPT</given-names> </name><name name-style="western"><surname>Thomas</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chandler</surname><given-names>J</given-names> </name><etal/></person-group><source>Cochrane Handbook for Systematic Reviews of Interventions Version 65 (Updated August 2024)</source><year>2024</year><publisher-name>Cochrane</publisher-name></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>ELY</given-names> </name><name name-style="western"><surname>Li</surname><given-names>JW</given-names> </name></person-group><article-title>Computer-aided quality control in colonoscopy: clinical applications and limitations</article-title><source>Clin Endosc</source><year>2025</year><month>12</month><day>17</day><pub-id pub-id-type="doi">10.5946/ce.2025.309</pub-id><pub-id pub-id-type="medline">41409023</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lwin</surname><given-names>WP</given-names> </name><name name-style="western"><surname>Ichimasa</surname><given-names>K</given-names> </name><name name-style="western"><surname>Kudo</surname><given-names>SE</given-names> </name><etal/></person-group><article-title>Clinical significance of computer-aided quality assessment systems in colonoscopy: a comprehensive review</article-title><source>Clin Endosc</source><year>2025</year><month>09</month><volume>58</volume><issue>5</issue><fpage>638</fpage><lpage>645</lpage><pub-id pub-id-type="doi">10.5946/ce.2025.022</pub-id><pub-id pub-id-type="medline">40438910</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Search strategy.</p><media xlink:href="jmir_v28i1e93618_app1.docx" xlink:title="DOCX File, 33 KB"/></supplementary-material><supplementary-material id="app2"><label>Checklist 1</label><p>PRISMA-Abstract checklist.</p><media xlink:href="jmir_v28i1e93618_app2.docx" xlink:title="DOCX File, 22 KB"/></supplementary-material><supplementary-material id="app3"><label>Checklist 2</label><p>PRISMA 2020 checklist.</p><media xlink:href="jmir_v28i1e93618_app3.docx" xlink:title="DOCX File, 223 KB"/></supplementary-material><supplementary-material id="app4"><label>Checklist 3</label><p>PRISMA-S checklist.</p><media xlink:href="jmir_v28i1e93618_app4.docx" xlink:title="DOCX File, 23 KB"/></supplementary-material></app-group></back></article>