<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e97588</article-id><article-id pub-id-type="doi">10.2196/97588</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Impact of Responsibility Allocation Structures on Diagnostic Quality in AI-Assisted Diagnosis: Randomized Controlled Experiment</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Liu</surname><given-names>Tianya</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1"/><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" corresp="yes" equal-contrib="yes"><name name-style="western"><surname>Wu</surname><given-names>Ji</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1"/><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib></contrib-group><aff id="aff1"><institution>School of Business, Sun Yat-sen University</institution><addr-line>Guangzhou</addr-line><addr-line>Guangdong</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Tsai</surname><given-names>Meng-Hsun</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Palama</surname><given-names>Valentina</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Ji Wu, PhD, School of Business, Sun Yat-sen University, Guangzhou, Guangdong, 510275, China, 86 13113620362; <email>wuji3@mail.sysu.edu.cn</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>all authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>23</day><month>7</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e97588</elocation-id><history><date date-type="received"><day>08</day><month>04</month><year>2026</year></date><date date-type="rev-recd"><day>29</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>29</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Tianya Liu, Ji Wu. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 23.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e97588"/><abstract><sec><title>Background</title><p>AI is increasingly used to support clinical diagnosis, but the appropriate allocation of responsibility between clinicians and AI remains unclear. Different responsibility structures may influence how clinicians evaluate AI recommendations and revise their diagnostic judgments.</p></sec><sec><title>Objective</title><p>This study aimed to examine how different physician-AI responsibility allocation structures affect diagnostic accuracy and confidence calibration during AI-assisted diagnosis.</p></sec><sec sec-type="methods"><title>Methods</title><p>This individually randomized, 4-arm, parallel-group controlled experiment was conducted in a simulated clinical environment on the Credamo platform (Beijing Yishumofa Technology Co, Ltd). A total of 105 licensed physicians were randomly assigned to the dynamic responsibility, full responsibility, equal responsibility, or control group. Nine participants who failed the prespecified attention checks were excluded from the primary analysis, resulting in an analytic sample of 96 physicians. Participants completed 10 clinical vignette&#x2013;based diagnostic tasks. The only between-group difference was the responsibility allocation structure. Primary outcomes were final diagnostic accuracy and confidence calibration; secondary outcomes included agreement rates and posttask subjective evaluations.</p></sec><sec sec-type="results"><title>Results</title><p>Responsibility allocation structures significantly modulated diagnostic quality. Compared to the control group, the full responsibility structure yielded no significant improvement in accuracy (mean 0.596, SD 0.152 vs 0.592, SD 0.169; mean difference 0.004, 95% CI &#x2212;0.089 to 0.098; <italic>P</italic>=.93) or confidence calibration (mean 0.193, SD 0.134 vs 0.150, SD 0.145; mean difference 0.043, 95% CI &#x2212;0.038 to 0.124; <italic>P</italic>=.29), while the equal responsibility structure showed suggestive evidence of lower diagnostic accuracy (mean 0.496, SD 0.185 vs 0.592, SD 0.169; mean difference &#x2212;0.096, 95% CI &#x2212;0.199 to 0.007; <italic>P</italic>=.07) and significantly poorer confidence calibration (mean 0.383, SD 0.175 vs 0.150, SD 0.145; mean difference 0.233, 95% CI 0.139 to 0.326; <italic>P</italic>&#x003C;.001). Conversely, the dynamic responsibility structure demonstrated superior performance, significantly improving diagnostic accuracy (mean 0.717, SD 0.105 vs 0.592, SD 0.169; mean difference 0.125, 95% CI 0.043 to 0.207; <italic>P</italic>=.004) and reducing confidence calibration (mean 0.040, SD 0.084 vs 0.150, SD 0.105; mean difference &#x2212;0.110, 95% CI &#x2212;0.179 to &#x2212;0.040; <italic>P</italic>=.003).</p></sec><sec sec-type="conclusions"><title>Conclusion</title><p>The dynamic responsibility structure may enable health care organizations to use AI more fully and appropriately without compromising clinicians&#x2019; diagnostic performance, thereby improving the safety and quality of AI-assisted diagnosis.</p></sec><sec><title>Trial Registration</title><p>ISRCTN Registry ISRCTN16943519; https://www.isrctn.com/ISRCTN16943519</p></sec></abstract><kwd-group><kwd>artificial intelligence</kwd><kwd>physician-AI collaboration</kwd><kwd>responsibility allocation</kwd><kwd>randomized controlled trial</kwd><kwd>diagnostic quality</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>The integration of AI into clinical workflows through decision support systems has transformed diagnostic and treatment pathways by providing real-time recommendations [<xref ref-type="bibr" rid="ref1">1</xref>]. However, the rapid adoption of these technologies introduces significant challenges regarding accountability [<xref ref-type="bibr" rid="ref2">2</xref>], particularly when an AI-assisted decision results in patient harm. Clinical decision-making in the age of AI is no longer a linear process; it is a complex interaction involving algorithm developers, health care organizations, and clinicians [<xref ref-type="bibr" rid="ref3">3</xref>]. Consequently, responsibility can no longer be viewed as a simple causal chain assigned to a single actor but must be understood as a distributed responsibility chain spanning multiple phases and stakeholders [<xref ref-type="bibr" rid="ref4">4</xref>].</p><p>This distributed structure has given rise to the responsibility gap, where the complexity and partial opacity of AI systems make it difficult to assign accountability under traditional legal and ethical mechanisms [<xref ref-type="bibr" rid="ref5">5</xref>]. Beyond the legal vacuum, a critical behavioral concern is the phenomenon of responsibility diffusion [<xref ref-type="bibr" rid="ref3">3</xref>]. Drawing on responsibility diffusion theory, the involvement of multiple actors may inadvertently diminish clinicians&#x2019; perceived personal accountability, potentially weakening their vigilance and critical evaluation before adopting AI-generated recommendations [<xref ref-type="bibr" rid="ref6">6</xref>]. In the high-stakes environment of health care, where diagnostic errors directly threaten patient safety, this potential reduction in vigilance is particularly hazardous.</p><p>The tension between human judgment and automated advice is further complicated by existing liability structures. While AI-related adverse outcomes often stem from multistage technical and organizational failures, current responsibility arrangements frequently place the primary burden of liability on the individual clinician [<xref ref-type="bibr" rid="ref7">7</xref>]. This concentration of liability may not only increase the perceived risk and uncertainty for physicians using AI but may also lead to a defensive decision-making posture that hinders effective human-AI collaboration [<xref ref-type="bibr" rid="ref8">8</xref>].</p><p>Despite extensive conceptual debate regarding the responsibility gap, there remains a critical lack of causal evidence demonstrating how specific responsibility allocation structures influence clinicians&#x2019; decision-making performance and collaborative behaviors in AI-assisted diagnosis. To address this gap, we conducted a 4-arm, parallel-group randomized controlled experiment involving 96 verified hospital physicians recruited through a professional platform. By simulating a clinical environment with validated vignettes, we compared the effects of full, equal, and dynamic responsibility allocation structures against a control condition. The primary objective of this study is to identify actionable and traceable responsibility design mechanisms to enhance diagnostic accuracy and confidence calibration, thereby providing a robust evidence base for the design of safe and accountable human-AI workflows in health care systems.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design and Setting</title><p>This study employed a preregistered [<xref ref-type="bibr" rid="ref9">9</xref>], 4-arm, parallel-group randomized controlled trial to investigate the causal impact of different responsibility allocation structures on clinicians&#x2019; diagnostic performance and collaborative behaviors [<xref ref-type="bibr" rid="ref10">10</xref>]. The experiment was conducted between December 2025 and January 2026 using a simulated clinical environment hosted on the Credamo platform (Beijing Yishumofa Technology Co, Ltd). This design allowed for the systematic comparison of 3 experimental responsibility allocation structures&#x2014;full, equal, and dynamic responsibility&#x2014;against a standard control condition during an AI-assisted diagnostic task.</p></sec><sec id="s2-2"><title>Participants and Recruitment</title><p>Participants were recruited through the Credamo platform and were required to be verified, licensed hospital clinicians capable of completing complex, vignette-based clinical decision-making tasks on digital devices. Upon providing electronic informed consent, participants were screened for eligibility and consistency. A total of 105 physicians were enrolled and randomly assigned to the 4 groups. Nine participants who failed the prespecified attention checks were excluded from the primary analysis, resulting in a final analytic sample of 96 physicians, with 24 participants in each group.</p></sec><sec id="s2-3"><title>Clinical Vignettes</title><p>Ten cases were randomly selected from the Chinese National Medical Licensing Examination question bank and adapted into clinical vignettes. Two clinical experts reviewed the case content and reference diagnoses and assessed the complexity and difficulty of each vignette; the final set comprised 5 easy and 5 difficult cases. A pilot test involving 20 licensed physicians was subsequently conducted to evaluate the clarity of the vignette descriptions, the comprehensibility of the experimental procedures, and the overall task difficulty; the experimental materials were then refined accordingly.</p></sec><sec id="s2-4"><title>Ethical Considerations</title><p>The study protocol was approved by the Ethics Committee of the School of Business, Sun Yat-sen University (approval number: BS20251227).</p></sec><sec id="s2-5"><title>Trial Registration</title><p>The study was preregistered on the AsPredicted platform (Wharton Credibility Lab) before data collection commenced. However, the trial was not prospectively registered in a clinical trial registry because we initially did not fully recognize that an online clinical vignette&#x2013;based simulation involving licensed physicians also required prospective trial registration. The trial was subsequently registered retrospectively with the ISRCTN registry (ISRCTN16943519). The study design, primary outcomes, exclusion criteria, and planned analyses had been specified before data collection and remained unchanged throughout the study. No outcomes were selectively omitted, added, or modified on the basis of the study findings.</p></sec><sec id="s2-6"><title>Randomization</title><p>Within a system-prespecified vignette-based clinical decision-making task, participants were randomly assigned in a 1:1:1:1 ratio using computer-generated simple randomization to 1 of 4 responsibility allocation structures: a no-intervention control condition and 3 experimental conditions manipulating different responsibility allocation structures [<xref ref-type="bibr" rid="ref11">11</xref>]. Allocation was implemented automatically by the platform back end, and the research team could neither foresee nor influence group assignment. As the intervention centered on the description and display of a responsibility allocation structure, participant blinding was not feasible; therefore, the trial used an open-label design. To minimize analytical bias, group labels were anonymized after data cleaning and coding, and statistical analyses were conducted by an analyst blinded to group identity [<xref ref-type="bibr" rid="ref12">12</xref>].</p></sec><sec id="s2-7"><title>Intervention</title><p>The experimental workflow, illustrated in <xref ref-type="fig" rid="figure1">Figure 1</xref>, began with participants acknowledging a group-specific responsibility allocation statement that defined the accountability boundaries between the clinicians and the AI algorithm developers [<xref ref-type="bibr" rid="ref13">13</xref>]. Participants then proceeded to the task interface where, after providing an initial diagnosis, they received diagnostic advice and rationale generated by ChatGPT (GPT-5.2). As the second component of the intervention, during each case response phase (before participants submitted their final diagnosis or revised their initial diagnosis), the system presented a group-specific adjustment prompt to reinforce the responsibility structure and guide final decision confirmation.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Experiment procedures. Currency conversion is based on the December 2025 average exchange rate (US $1=CN &#x00A5;7.0432).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e97588_fig01.png"/></fig></sec><sec id="s2-8"><title>Procedures</title><p>After entering the online experiment platform, participants first completed a personal information form and read the responsibility statement displayed by the system. They then completed diagnostic tasks for 10 clinical vignettes. For each case, participants made an initial diagnosis based on the case information and rated their confidence. Then, the system displayed the AI diagnostic conclusion and asked participants to rate agreement with the AI recommendation and to record human-AI outcome consistency. If the 2 conclusions were consistent, participants proceeded directly to the next case; otherwise, the system asked whether they wished to revise their initial diagnosis. After revising, participants rerated their confidence and submitted the final diagnosis for that case. After completing all cases, participants completed a posttask questionnaire assessing outcomes [<xref ref-type="bibr" rid="ref14">14</xref>].</p></sec><sec id="s2-9"><title>Outcome Measures</title><p>Before the diagnostic tasks, participants reported their sociodemographic and professional characteristics, previous experience with AI, and trust in AI.</p><p>The postintervention outcomes included diagnostic quality, agreement, adjustment, and posttask subjective measures. Diagnostic quality comprised initial and final diagnostic accuracy, each coded dichotomously against the case-specific reference standard (1=correct diagnosis, 0=otherwise). We also calculated calibration for the final diagnosis to assess the concordance between confidence and actual correctness: confidence in the final diagnosis was rated on a 1 to 7 scale and linearly rescaled to a 0 to 1 range; accuracy was coded as 1 for correct and 0 for incorrect. Calibration was defined as the absolute difference between confidence and accuracy, with lower values indicating better calibration. Agreement captured whether participants endorsed key elements of the AI reasoning and was coded as a binary variable (1=agree, 0=otherwise). Adjustment reflected whether participants modified their diagnostic conclusion in response to the AI advice and was also coded dichotomously (1=adjusted, 0=otherwise). After completing the task, participants reported their subjective perceptions of collaborative efficacy, cognitive load, and credit attribution during the task using Likert-type scales, each scored from 1 to 7 (1=strongly disagree, 7=strongly agree).</p></sec><sec id="s2-10"><title>Data Analysis</title><p>Quantitative analyses were performed using Stata (version 15.1; StataCorp). We conducted group comparisons for primary and secondary outcomes, including diagnostic quality measures and posttask subjective scales. All tests were 2-sided. Results with <italic>P</italic>&#x003C;.05 were considered to meet the conventional threshold for statistical significance, whereas those with <italic>P</italic>&#x003C;.05-.10 were reported as suggestive evidence and were not interpreted as conclusive findings.</p></sec><sec id="s2-11"><title>Patient and Public Involvement</title><p>Patients and the public were not involved in the design, conduct, reporting, or dissemination plans of this research. The study involved licensed clinicians completing simulated vignette-based diagnostic tasks.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Characteristics of Participants</title><p>A total of 96 verified hospital physicians were successfully recruited and underwent 1:1:1:1 randomization into the 4 study arms (n=24 per group; <xref ref-type="fig" rid="figure2">Figure 2</xref>). Data from all 96 participants were included in the final primary analysis after passing internal consistency checks. The sample was balanced across genders (51/96, 53.1% female) and primarily composed of clinicians in early- to midcareer stages, with 87.5% (n=84) aged between 21 and 40 years. In the sample, 42.7% (n=41) held intermediate professional titles, and clinical experience was evenly distributed, with 51.1% (n=49) of the cohort having more than 6 years of practice. AI use experience was high, with 79.2% (n=76) of participants having used AI tools for 6 months or longer.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Flow diagram.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e97588_fig02.png"/></fig></sec><sec id="s3-2"><title>Primary Outcomes</title><p>Responsibility allocation structures were significantly associated with overall diagnostic quality. The primary results, including point estimates and 95% CIs derived from the experimental data, are presented in <xref ref-type="table" rid="table1">Tables 1</xref> and <xref ref-type="table" rid="table2">2</xref>.</p><p>Compared to the control group, the full responsibility structure did not yield a statistically significant improvement in final diagnostic accuracy (mean 0.596, SD 0.152 vs mean 0.592, SD 0.169; difference 0.004, 95% CI &#x2013;0.089 to 0.098; <italic>P</italic>=.93) or confidence calibration (mean 0.193, SD 0.134 vs mean 0.150, SD 0.145; difference 0.043, 95% CI &#x2013;0.038 to 0.124; <italic>P</italic>=.29). The equal responsibility structure suggested a reduction in accuracy, but this was not statistically significant (mean 0.496, SD 0.185 vs mean 0.592, SD 0.169; difference &#x2013;0.096, 95% CI &#x2013;0.199 to 0.007; <italic>P</italic>=.07); furthermore, this structure resulted in significantly poorer confidence calibration compared to the control group (mean 0.383, SD 0.175 vs mean 0.150, SD 0.145; difference 0.233, 95% CI 0.139 to 0.326; <italic>P</italic>&#x003C;.001), indicating a heightened mismatch between clinician confidence and objective accuracy.</p><p>In contrast, the dynamic responsibility structure demonstrated a robust advantage, significantly improving final diagnostic accuracy (mean 0.717, SD 0.105 vs mean 0.592, SD 0.169; difference 0.125, 95% CI 0.043 to 0.207; <italic>P</italic>=.004) and reducing confidence calibration (mean 0.040, SD 0.084 vs mean 0.150, SD 0.145; difference &#x2013;0.110, 95% CI &#x2013;0.179 to &#x2013;0.040; <italic>P</italic>=.003). Pairwise comparisons further confirmed that the dynamic structure significantly outperformed both the full (accuracy <italic>P=</italic>.003; calibration <italic>P&#x003C;</italic>.001) and equal responsibility models (accuracy <italic>P</italic>&#x003C;.001; calibration <italic>P</italic>&#x003C;.001), suggesting that embedding responsibility cues at points of human-AI disagreement promotes superior diagnostic verification.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Diagnostic accuracy comparisons across responsibility allocation structures.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Comparison</td><td align="left" valign="bottom">Group mean (SD)</td><td align="left" valign="bottom"><italic>t</italic> test (<italic>df</italic>)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Full responsibility vs control</td><td align="left" valign="top">0.596 (0.152) vs 0.592 (0.169)</td><td align="left" valign="top">0.09 (45.47)</td><td align="left" valign="top">.93</td></tr><tr><td align="left" valign="top">Equal responsibility vs control</td><td align="left" valign="top">0.496 (0.185) vs 0.592 (0.169)</td><td align="left" valign="top">&#x2013;1.87 (45.63)</td><td align="left" valign="top">.07</td></tr><tr><td align="left" valign="top">Dynamic responsibility vs control</td><td align="left" valign="top">0.717 (0.105) vs 0.592 (0.169)</td><td align="left" valign="top">3.08 (38.42)</td><td align="left" valign="top">.004</td></tr><tr><td align="left" valign="top">Full vs equal responsibility</td><td align="left" valign="top">0.596 (0.152) vs 0.496 (0.185)</td><td align="left" valign="top">2.05 (44.28)</td><td align="left" valign="top">.047</td></tr><tr><td align="left" valign="top">Full vs dynamic responsibility</td><td align="left" valign="top">0.596 (0.152) vs 0.717 (0.105)</td><td align="left" valign="top">&#x2013;3.21 (40.91)</td><td align="left" valign="top">.003</td></tr><tr><td align="left" valign="top">Equal vs dynamic responsibility</td><td align="left" valign="top">0.496 (0.185) vs 0.717 (0.105)</td><td align="left" valign="top">&#x2013;5.08 (36.38)</td><td align="left" valign="top">&#x003C;.001</td></tr></tbody></table></table-wrap><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Diagnostic calibration comparisons across responsibility allocation structures. Note: Lower confidence calibration values indicate better calibration.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Comparison</td><td align="left" valign="bottom">Group mean (SD)</td><td align="left" valign="bottom"><italic>t</italic> test (<italic>df</italic>)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Full responsibility vs control</td><td align="left" valign="top">0.193 (0.134) vs 0.150 (0.145)</td><td align="left" valign="top">1.08 (45.71)</td><td align="left" valign="top">.29</td></tr><tr><td align="left" valign="top">Equal responsibility vs control</td><td align="left" valign="top">0.383 (0.175) vs 0.150 (0.145)</td><td align="left" valign="top">5.01 (44.40</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Dynamic responsibility vs control</td><td align="left" valign="top">0.040 (0.084) vs 0.150 (0.145)</td><td align="left" valign="top">&#x2013;3.20 (36.94)</td><td align="left" valign="top">.003</td></tr><tr><td align="left" valign="top">Full vs equal responsibility</td><td align="left" valign="top">0.193 (0.134) vs 0.383 (0.175)</td><td align="left" valign="top">&#x2013;4.21 (42.97)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Full vs dynamic responsibility</td><td align="left" valign="top">0.193 (0.134) vs 0.040 (0.084)</td><td align="left" valign="top">4.75 (38.76)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Equal vs dynamic responsibility</td><td align="left" valign="top">0.383 (0.175) vs 0.040 (0.084)</td><td align="left" valign="top">8.62 (33.05)</td><td align="left" valign="top">&#x003C;.001</td></tr></tbody></table></table-wrap></sec><sec id="s3-3"><title>Secondary Outcomes</title><p>Secondary outcomes are presented in <xref ref-type="fig" rid="figure3">Figure 3</xref>. For initial accuracy, no statistically significant differences were observed across responsibility allocation structures, indicating that the responsibility structures did not affect initial judgments. Compared with the control condition and the full responsibility structure, both the equal responsibility and dynamic responsibility structures showed higher levels of agreement and adjustment, although the difference between these 2 structures was not significant.</p><p>Regarding subjective evaluations, the equal responsibility and dynamic responsibility structures were associated with lower cognitive load, higher collaborative effectiveness, and lower credit attribution, suggesting that under these 2 responsibility allocation structures, clinicians were less likely to attribute diagnostic contributions entirely to themselves while experiencing a lighter task burden and a more favorable sense of collaboration.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Secondary results.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e97588_fig03.png"/></fig></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>The primary objective of this randomized controlled experiment was to evaluate how various responsibility allocation structures influence the diagnostic performance of clinicians when assisted by AI. Our findings indicate that responsibility structures may play an important role in shaping diagnostic quality. The principal finding is that a dynamic responsibility structure, in which accountability cues are specifically triggered during human-AI disagreement, significantly enhances both diagnostic accuracy and confidence calibration compared to the full or equal responsibility models. Conversely, the equal responsibility framework, while appearing collaborative, was associated with poorer confidence calibration, a pattern that may be consistent with responsibility diffusion.</p></sec><sec id="s4-2"><title>Interpretation and Implications</title><p>Our findings align with and extend the existing literature on automation bias and the responsibility gap in health care. Some previous research has highlighted that clinicians may overrely on automated advice, particularly when liability is poorly defined. While some scholars have argued that placing full liability on clinicians is necessary to ensure safety, our data suggest that this sole accountability model (full responsibility) does not significantly improve diagnostic accuracy or calibration compared to the control. This contradicts the traditional &#x201C;captain of the ship&#x201D; legal doctrine, suggesting that liability pressure alone may be insufficient to overcome cognitive biases [<xref ref-type="bibr" rid="ref15">15</xref>].</p><p>Our observation regarding the equal responsibility group may be consistent with research in social psychology concerning social loafing and responsibility diffusion. When accountability is shared equally, clinicians may distribute part of the cognitive burden to the AI, which may help explain the observed poorer confidence calibration [<xref ref-type="bibr" rid="ref16">16</xref>]. However, this mechanism was not directly measured in the present study. The superior performance of the dynamic responsibility model suggests that a human factors engineering approach that prompts clinicians at the moment of decision conflict may support diagnostic verification more effectively than static pretask disclosures [<xref ref-type="bibr" rid="ref17">17</xref>].</p><p>For clinical governance and the design of health IT systems, this study offers potential implications. Current regulatory frameworks often treat responsibility as a static legal assignment. Our results suggest that AI-assisted workflows could consider incorporating accountability by design [<xref ref-type="bibr" rid="ref18">18</xref>]. By embedding responsibility cues within the decision-making interface, specifically when the clinician and the AI disagree, system designers may help reduce the risk of automation bias and promote more rigorous cross-checking behavior. For policymakers, this suggests that the distribution of liability may not be appropriately conceptualized as a fixed binary (human vs machine) but as a dynamic framework that encourages active human oversight.</p></sec><sec id="s4-3"><title>Limitations</title><p>Several limitations should be acknowledged. First, participants were aware of the responsibility structure to which they were assigned, which may have influenced their diagnostic revision behavior. Second, the vignette-based online simulation used in this study could not fully reproduce the decision pressures encountered in real-world clinical settings, which may limit the generalizability of the findings to clinical practice. Third, perceived responsibility and responsibility diffusion were not directly measured. Accordingly, responsibility diffusion should be interpreted as a plausible explanation rather than an empirically established causal mechanism.</p><p>Future research could measure perceived responsibility and related constructs to examine whether these constructs help explain the relationship between responsibility allocation structures and diagnostic performance. Building on this, studies should further evaluate the effects of different responsibility allocation structures in settings involving more realistic decision pressures and should investigate the long-term effects of dynamic responsibility structures. It is possible that alert fatigue could diminish the effectiveness of dynamic cues over time. Additionally, studies should explore how different levels of AI transparency interact with responsibility structures. For instance, would a black-box AI require more aggressive responsibility framing than a transparent one to achieve a comparable level of clinician vigilance? Exploring these interactions may contribute to the development of robust safety standards for AI integration in medicine.</p></sec><sec id="s4-4"><title>Conclusions</title><p>In summary, this study provides causal evidence that responsibility allocation structures can shape clinicians&#x2019; performance in AI-assisted clinical decision-making. The safe and effective use of clinical AI depends not only on algorithmic performance but also on whether human responsibility is consistently reinforced through organizational arrangements and clinical workflows. Compared with static responsibility arrangements, dynamic responsibility cues delivered at key decision points may better support clinicians&#x2019; active oversight and encourage more careful scrutiny of AI recommendations. These findings suggest that clinical AI governance should address not only technical reliability but also responsibility mechanisms and the risks associated with human-AI collaboration, thereby supporting safer and more trustworthy clinical use.</p></sec></sec></body><back><ack><p>The authors thank all physicians who participated in this study for their time and effort. Editorial assistance for English language and style was supported by generative AI tools, including ChatGPT (OpenAI), which were used solely for language editing, formatting, and clarification of reviewer comments during manuscript preparation. All AI-assisted content was carefully reviewed and verified by the authors, who are solely responsible for the content, scientific accuracy, and conclusions of the final manuscript.</p></ack><notes><sec><title>Funding</title><p>This research was funded by the National Natural Science Foundation of China (72322020, 72071218) and the Guangdong Basic and Applied Basic Research Foundation (2023B151020073).</p></sec><sec><title>Data Availability</title><p>Data are available upon reasonable request.</p></sec></notes><fn-group><fn fn-type="con"><p>TL contributed to conceptualization, methodology, investigation, formal analysis, data curation, and writing of the original draft. JW contributed to conceptualization, supervision, interpretation of the results, and review and editing of the manuscript. Both authors read and approved the final manuscript.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn><fn fn-type="other"><p><bold>Editorial Notice</bold></p><p>This randomized study was retrospectively registered, explained by the authors as follows: "The study was preregistered on the AsPredicted platform before data collection commenced. However, the trial was not prospectively registered in a clinical trial registry because we initially did not fully recognize that an online clinical vignette&#x2013;based simulation involving licensed physicians also required prospective trial registration. The trial was subsequently registered retrospectively with the ISRCTN registry (ISRCTN16943519). The study design, primary outcomes, exclusion criteria, and planned analyses had been specified before data collection and remained unchanged throughout the study. No outcomes were selectively omitted, added, or modified on the basis of the study findings." The editor granted an exception from ICMJE rules mandating prospective registration of randomized trials, because the risk of bias appears low. However, readers are advised to carefully assess the validity of any potential explicit or implicit claims related to primary outcomes or effectiveness, as retrospective registration does not prevent authors from changing their outcome measures retrospectively.</p></fn></fn-group><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Choi</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>G</given-names> </name></person-group><article-title>AI as a collaborator: the impact of generative AI collaboration on users&#x2019; subjective effort and responsibility perceptions [Article in Korean]</article-title><source>J Inf Syst</source><year>2025</year><volume>34</volume><issue>1</issue><fpage>125</fpage><lpage>153</lpage><pub-id pub-id-type="doi">10.5859/KAIS.2025.34.1.125</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Simmler</surname><given-names>M</given-names> </name></person-group><article-title>Responsibility gap or responsibility shift? The attribution of criminal responsibility in human&#x2013;machine interaction</article-title><source>Inf Commun Soc</source><year>2024</year><month>04</month><day>25</day><volume>27</volume><issue>6</issue><fpage>1142</fpage><lpage>1162</lpage><pub-id pub-id-type="doi">10.1080/1369118X.2023.2239895</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Danaher</surname><given-names>J</given-names> </name></person-group><article-title>Robots, law and the retribution gap</article-title><source>Ethics Inf Technol</source><year>2016</year><month>12</month><volume>18</volume><issue>4</issue><fpage>299</fpage><lpage>309</lpage><pub-id pub-id-type="doi">10.1007/s10676-016-9403-3</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Veluwenkamp</surname><given-names>H</given-names> </name></person-group><article-title>What responsibility gaps are and what they should be</article-title><source>Ethics Inf Technol</source><year>2025</year><month>03</month><volume>27</volume><issue>1</issue><fpage>1</fpage><lpage>13</lpage><pub-id pub-id-type="doi">10.1007/s10676-025-09823-8</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bleher</surname><given-names>H</given-names> </name><name name-style="western"><surname>Braun</surname><given-names>M</given-names> </name></person-group><article-title>Diffused responsibility: attributions of responsibility in the use of AI-driven clinical decision support systems</article-title><source>AI Ethics</source><year>2022</year><volume>2</volume><issue>4</issue><fpage>747</fpage><lpage>761</lpage><pub-id pub-id-type="doi">10.1007/s43681-022-00135-x</pub-id><pub-id pub-id-type="medline">35098247</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Azad</surname><given-names>A</given-names> </name></person-group><article-title>243&#x2005;Leveraging AI to enhance patient safety and its associated implications</article-title><source>BMJ Open Qual</source><year>2025</year><month>05</month><volume>14</volume><issue>Suppl 3</issue><fpage>A183</fpage><lpage>A184</lpage><pub-id pub-id-type="doi">10.1136/bmjoq-2025-QSHU.243</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mittelstadt</surname><given-names>B</given-names> </name></person-group><article-title>Principles alone cannot guarantee ethical AI</article-title><source>Nat Mach Intell</source><year>2019</year><volume>1</volume><issue>11</issue><fpage>501</fpage><lpage>507</lpage><pub-id pub-id-type="doi">10.1038/s42256-019-0114-4</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Agley</surname><given-names>J</given-names> </name><name name-style="western"><surname>Henderson</surname><given-names>C</given-names> </name><name name-style="western"><surname>Nair</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Recruiting a national sample of first response agencies to participate in an overdose prevention research project: randomized controlled trial and feasibility study</article-title><source>JMIR Form Res</source><year>2026</year><month>04</month><day>2</day><volume>10</volume><fpage>e81743</fpage><pub-id pub-id-type="doi">10.2196/81743</pub-id><pub-id pub-id-type="medline">41925721</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>T</given-names> </name></person-group><article-title>Diagnostic quality of dynamic responsibility groups in clinical decision-making</article-title><source>AsPredicted preregistration record</source><access-date>2025-12-27</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://aspredicted.org/b5ib3v.pdf">https://aspredicted.org/b5ib3v.pdf</ext-link></comment></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>M</given-names> </name><name name-style="western"><surname>Guo</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>A gamified mobile health intervention to promote physical activity, executive function, and mental health in college students: randomized controlled trial</article-title><source>J Med Internet Res</source><year>2026</year><month>04</month><day>7</day><volume>28</volume><fpage>e82769</fpage><pub-id pub-id-type="doi">10.2196/82769</pub-id><pub-id pub-id-type="medline">41945642</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Arnal-Vall&#x00E9;s</surname><given-names>MP</given-names> </name><name name-style="western"><surname>Soto-Ruiz</surname><given-names>N</given-names> </name><name name-style="western"><surname>Bays-Moneo</surname><given-names>AB</given-names> </name><name name-style="western"><surname>Garc&#x00ED;a-Vivar</surname><given-names>C</given-names> </name><name name-style="western"><surname>Escalada-Hern&#x00E1;ndez</surname><given-names>P</given-names> </name></person-group><article-title>Motor imagery and action observation in breast cancer survivors: protocol for a randomized controlled trial</article-title><source>JMIR Res Protoc</source><year>2026</year><month>03</month><day>30</day><volume>15</volume><fpage>e85469</fpage><pub-id pub-id-type="doi">10.2196/85469</pub-id><pub-id pub-id-type="medline">41911468</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Scott</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lurgain</surname><given-names>JG</given-names> </name><name name-style="western"><surname>Day</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Randomised controlled community trial assessing efficacy of the AWACAN-ED public toolkit to improve cancer symptom awareness and intention to seek help in South Africa and Zimbabwe: study protocol</article-title><source>BMJ Open</source><year>2026</year><month>01</month><day>14</day><volume>16</volume><issue>1</issue><fpage>e106400</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2025-106400</pub-id><pub-id pub-id-type="medline">41535080</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Verdicchio</surname><given-names>M</given-names> </name><name name-style="western"><surname>Perin</surname><given-names>A</given-names> </name></person-group><article-title>When doctors and AI interact: on human responsibility for artificial risks</article-title><source>Philos Technol</source><year>2022</year><volume>35</volume><issue>1</issue><fpage>11</fpage><pub-id pub-id-type="doi">10.1007/s13347-022-00506-6</pub-id><pub-id pub-id-type="medline">35223383</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jayaram</surname><given-names>M</given-names> </name><name name-style="western"><surname>Adams</surname><given-names>CE</given-names> </name><name name-style="western"><surname>Friedel</surname><given-names>JS</given-names> </name><etal/></person-group><article-title>Day of the week to tweet: a randomised controlled trial</article-title><source>BMJ Open</source><year>2019</year><month>04</month><day>4</day><volume>9</volume><issue>4</issue><fpage>e025380</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2018-025380</pub-id><pub-id pub-id-type="medline">30948581</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dai</surname><given-names>T</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>S</given-names> </name></person-group><article-title>Artificial intelligence on call: the physician&#x2019;s decision of whether to use AI in clinical practice</article-title><source>J Mark Res</source><year>2025</year><month>10</month><volume>62</volume><issue>5</issue><fpage>854</fpage><lpage>875</lpage><pub-id pub-id-type="doi">10.1177/00222437251332898</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cooper</surname><given-names>DA</given-names> </name></person-group><article-title>Diffusion of responsibility for actions with advice</article-title><source>J Behav Decis Mak</source><year>2024</year><month>10</month><volume>37</volume><issue>4</issue><fpage>e2415</fpage><pub-id pub-id-type="doi">10.1002/bdm.2415</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Naik</surname><given-names>N</given-names> </name><name name-style="western"><surname>Hameed</surname><given-names>BMZ</given-names> </name><name name-style="western"><surname>Shetty</surname><given-names>DK</given-names> </name><etal/></person-group><article-title>Legal and ethical consideration in artificial intelligence in healthcare: who takes responsibility?</article-title><source>Front Surg</source><year>2022</year><volume>9</volume><fpage>862322</fpage><pub-id pub-id-type="doi">10.3389/fsurg.2022.862322</pub-id><pub-id pub-id-type="medline">35360424</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dullabh</surname><given-names>P</given-names> </name><name name-style="western"><surname>Zott</surname><given-names>C</given-names> </name><name name-style="western"><surname>Gauthreaux</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Integrating generative AI into patient-centered clinical decision support: viewpoint on research and practice considerations</article-title><source>J Med Internet Res</source><year>2026</year><month>04</month><day>1</day><volume>28</volume><fpage>e81628</fpage><pub-id pub-id-type="doi">10.2196/81628</pub-id><pub-id pub-id-type="medline">41921087</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Checklist 1</label><p>CONSORT checklist.</p><media xlink:href="jmir_v28i1e97588_app1.pdf" xlink:title="PDF File, 278 KB"/></supplementary-material></app-group></back></article>