<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e91472</article-id><article-id pub-id-type="doi">10.2196/91472</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Ethics of Autonomous AI Clinical Trials: Delphi Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Nichol</surname><given-names>Ariadne A</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Youssef</surname><given-names>Alaa</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Larson</surname><given-names>David B</given-names></name><degrees>MBA, MD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Abramoff</surname><given-names>Michael</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wolf</surname><given-names>Risa M</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Char</surname><given-names>Danton</given-names></name><degrees>MS, MD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Martinez-Martin</surname><given-names>Nicole</given-names></name><degrees>JD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff6">6</xref></contrib></contrib-group><aff id="aff1"><institution>Center for Biomedical Ethics, Stanford Medicine</institution><addr-line>300 Pasteur Drive</addr-line><addr-line>Stanford</addr-line><addr-line>CA</addr-line><country>United States</country></aff><aff id="aff2"><institution>Department of Radiology, Stanford Medicine</institution><addr-line>Stanford</addr-line><addr-line>CA</addr-line><country>United States</country></aff><aff id="aff3"><institution>Department of Ophthalmology and Visual Sciences, University of Iowa Health Care</institution><addr-line>Iowa City</addr-line><addr-line>IA</addr-line><country>United States</country></aff><aff id="aff4"><institution>Department of Pediatrics, Division of Endocrinology, Johns Hopkins Medicine</institution><addr-line>Baltimore</addr-line><addr-line>MD</addr-line><country>United States</country></aff><aff id="aff5"><institution>Department of Anesthesiology, Division of Pediatric Cardiac Anesthesia, Stanford Medicine</institution><addr-line>Stanford</addr-line><addr-line>CA</addr-line><country>United States</country></aff><aff id="aff6"><institution>Department of Psychiatry and Behavioral Sciences, Stanford Medicine</institution><addr-line>Stanford</addr-line><addr-line>CA</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Law</surname><given-names>Stephanie</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Satasiya</surname><given-names>Kinjal</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Gupta</surname><given-names>Sandeep</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Nicole Martinez-Martin, JD, PhD, Center for Biomedical Ethics, Stanford Medicine, 300 Pasteur Drive, Stanford, CA, 94305, United States, 1 (650) 723-4480; <email>nicolemz@stanford.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>21</day><month>8</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e91472</elocation-id><history><date date-type="received"><day>15</day><month>01</month><year>2026</year></date><date date-type="rev-recd"><day>25</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>25</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Ariadne A Nichol, Alaa Youssef, David B Larson, Michael Abramoff, Risa M Wolf, Danton Char, Nicole Martinez-Martin. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 21.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e91472"/><abstract><sec><title>Background</title><p>Safe implementation of autonomous AI in medicine requires rigorous evaluation through clinical trials. The 7 guiding principles for ethical clinical research endorsed by the National Institutes of Health (NIH) provide an established framework for promoting scientific rigor and protecting the safety of human participants. However, clinical trials of autonomous AI raise novel ethical issues that require adaptation of these principles to account for effects that can vary across stakeholders and implementation contexts, including model performance across clinical settings. Incorporating expert perspectives on such challenges is critical to developing effective and ethically robust guidelines for autonomous AI clinical trials.</p></sec><sec><title>Objective</title><p>This Delphi study aimed to generate expert consensus on how the National Institutes of Health&#x2019;s 7 principles of ethical clinical research should be applied to clinical trials of autonomous AI.</p></sec><sec sec-type="methods"><title>Methods</title><p>We conducted 2 rounds of surveys followed by a final virtual meeting using a modified Delphi approach with a multidisciplinary expert panel. Participants were purposively identified through PubMed literature searches and selected for expertise in AI, data science, ophthalmology, public policy, law, bioethics, and patient advocacy. Round 1 used open-ended survey questions based on a vignette describing an autonomous AI tool. Round 1 survey responses were analyzed qualitatively using thematic coding, and were used to generate representative statements for Round 2. In round 2, panelists rated statements on a 5-point Likert scale. In accordance with Delphi methodology, statements of consensus within the surveys (at least 80% rating agreement) and moderate agreement (60%&#x2010;80% rating agreement) were identified for further discussion and iteration. Findings from a final virtual meeting were then analyzed thematically and synthesized into actionable recommendations.</p></sec><sec sec-type="results"><title>Results</title><p>Fourteen expert panelists participated in the Delphi study over a 6-month period. Participation was 12 (85.7%) of 14 experts in round 1, 10 (71.4%) of 14 experts in round 2, and 13 (92.9%) of 14 experts in the final virtual meeting. Round 2 survey results yielded 9 strong agreement statements, 2 moderate agreement statements, and 4 divisive statements for participants to explore further in developing recommendations. Final recommendations from the virtual meeting addressed transparency regarding training and validation data, bias assessment before deployment, performance across clinical settings, health inequities before implementation, stakeholder engagement, informed consent, comparison of AI tools with existing standards of care, downstream access to care after AI-generated recommendations, and cost and access implications.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Ethical evaluation of autonomous AI clinical trials should extend beyond technical accuracy to help mitigate potential harms to patients. This study highlights key ethical considerations for clinical trials of autonomous AI and provides consensus recommendations from a multidisciplinary Delphi panel. These recommendations can inform future research, policy, and guidance for the ethical development and implementation of autonomous AI clinical trials.</p></sec></abstract><kwd-group><kwd>autonomous AI</kwd><kwd>clinical trial</kwd><kwd>ethics</kwd><kwd>Delphi</kwd><kwd>transparency</kwd><kwd>bioethics</kwd><kwd>public policy</kwd><kwd>consensus</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Autonomous AI is increasingly being evaluated for use in health care, raising important questions for the ethics of clinical research. Unlike AI-based decision support tools meant to assist clinicians, autonomous AI tools are designed to screen for disease or provide clinical recommendation outputs without specialist oversight at the point of care. As these tools move from development into clinical research trials, regulators must determine how established research ethics principles should apply to such novel technology.</p><p>This question has become more urgent, as AI-based medical devices have rapidly expanded. Since the first autonomous AI system received the US Food and Drug Administration (FDA) De Novo authorization in 2018, the number of AI medical devices authorized has grown rapidly, now exceeding 1000 devices [<xref ref-type="bibr" rid="ref1">1</xref>]. Yet, clinical evaluation of AI tools in health care remains challenging [<xref ref-type="bibr" rid="ref2">2</xref>-<xref ref-type="bibr" rid="ref5">5</xref>]. Systematic reviews have highlighted significant limitations, including the lack of clinically relevant endpoints and a high risk of bias [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. Harms from AI in health care, particularly bias and differing performance across populations, have often been discovered only after deployment [<xref ref-type="bibr" rid="ref8">8</xref>-<xref ref-type="bibr" rid="ref10">10</xref>]. Ethics principles for medical research provide guidance for assuring scientific benefit as well as protection of human subjects. There is a need for more explicit guidance on how ethical principles should be applied to the evaluation of clinical AI.</p><p>Current guidance around ethics of AI in health care commonly emphasizes accountability, transparency, bias mitigation, and governance. For example, the World Health Organization&#x2019;s AI for Health 2024 guidance frames AI ethics in terms of design, deployment, use, and governance, and the International Medical Device Regulators Forum has guiding 2025 principles for good machine learning practice, which include transparency [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref12">12</xref>]. At a national level, the FDA has guidance from 2025 centering around transparency and risk-based evaluation of intended use of AI-based medical devices [<xref ref-type="bibr" rid="ref13">13</xref>]. However, a key gap remains in that existing guidance broadly promotes responsible development but provides less direction for how clinical trials of autonomous AI should be evaluated as human subjects research.</p><p>Clinical research has long been guided by 7 core ethical principles delineated by Emanuel et al and endorsed by the National Institutes of Health (NIH): social and clinical value, scientific validity, fair participant selection, favorable risk-benefit ratio, independent review, informed consent, and respect for human participants [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. These ethical research principles have been described as &#x201C;universal&#x201D; and are to be adapted to the &#x201C;health, economic, cultural, and technological conditions&#x201D; in which research is conducted [<xref ref-type="bibr" rid="ref14">14</xref>]. These principles promote ethical research and guide how the scientific and social benefit of research is evaluated. Certain features of AI tools add complexity to applying these principles: the impact on multiple stakeholders (eg, developers, clinicians, and patients); the impact of deployment in different settings or populations on performance; and/or the need to account for stakeholder values throughout the process of design to deployment of AI in health care. Ethical dilemmas are likely to emerge at these tension points, where there are perceived trade-offs between different values and goals [<xref ref-type="bibr" rid="ref16">16</xref>]. For example, a study of an AI-based chest x-ray prediction model that underdiagnosed specific subpopulations (eg, Black, Hispanic, and Medicaid subpopulations) demonstrated that there can be trade-offs between how an algorithm may acceptably account for certain demographic characteristics (eg, sex, race, ethnicity, and insurance status) versus safety for patients in general [<xref ref-type="bibr" rid="ref17">17</xref>]. This kind of trade-off creates challenges for applying the NIH principles.</p><p>To ground these issues in a concrete case, this Delphi study used an autonomous AI screening tool for diabetic retinopathy (DR) as an exemplar. DR is a leading cause of blindness worldwide [<xref ref-type="bibr" rid="ref18">18</xref>]. In 2018, the FDA granted the first De Novo authorization for an autonomous AI tool, which was for screening for DR, and it was then implemented in youth populations [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref21">21</xref>]. This case provided expert panelists with a specific example of an autonomous AI-based screening tool that has undergone FDA review and was evaluated in a clinical context.</p><p>The aim of this Delphi study was to identify areas of expert consensus regarding how the 7 NIH ethical principles should apply to clinical trials of autonomous AI tools. Further characterization of how to conduct ethical clinical research with AI tools is urgently needed to support safe implementation.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Overview</title><p>We convened a Delphi panel to understand the diverse experts&#x2019; views about the ethical considerations arising with a clinical trial of an autonomous AI that screened for DR in pediatric patients. The Delphi methodology was used because it is an established, structured, and iterative process that involves a panel of relevant experts to identify areas of consensus and dissent regarding an issue [<xref ref-type="bibr" rid="ref22">22</xref>-<xref ref-type="bibr" rid="ref24">24</xref>]. We used a modified Delphi approach, conducted in 3 rounds (<xref ref-type="fig" rid="figure1">Figure 1</xref>): a first-round survey; a second-round survey incorporating summary results from the first round; and a final virtual meeting round in which the expert panel reviewed consensus statements and further developed recommendations. See <xref ref-type="supplementary-material" rid="app1">Checklist 1</xref> for the DELPHISTAR reporting guidelines checklist.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Delphi process. NIH: National Institutes of Health.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e91472_fig01.png"/></fig></sec><sec id="s2-2"><title>Composition of the Delphi Panel</title><p>Delphi panels can range in size from 10 participants to larger groups. We aimed for a panel of 12 to 15 participants. Delphi studies in the health sciences involve an average of 15 participants [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>]. This number is thought to balance the needed range of expertise and a panel size suitable for engaged conversation for the final round meeting [<xref ref-type="bibr" rid="ref24">24</xref>]. We identified potential Delphi participants by searching the biomedical literature in the PubMed database to identify experts in areas relevant to questions of AI use in retinopathy, including in AI and data science, ophthalmology, public health policy, law, bioethics, and patient advocacy (<xref ref-type="table" rid="table1">Table 1</xref>). Several potential participants were selected for each category of expertise and ranked based on factors such as years of experience, previous work on similar projects, and publications. Email invitations were sent based on the rankings to fill each relevant area of expertise. Twenty-four invitations were sent out in total. Ten potential participants did not respond or declined, citing time constraints. Fourteen experts agreed to participate and represented the necessary categories of expertise. Several participants had overlapping relevant areas of expertise. For example, 2 of the 3 individuals with bioethics expertise also represented expertise in AI and clinical ophthalmology. Participants were given information regarding the Delphi study and consented to participate in the study. Key considerations when selecting panel size include ensuring a diverse range of expert perspectives, timing in administration of iterative rounds of surveys, and coordination of scheduling a final group discussion with multiple experts.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Characteristics of Delphi participants.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Characteristics</td><td align="left" valign="bottom">Values (N=14), n (%)</td></tr></thead><tbody><tr><td align="left" valign="top">Sex</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Male</td><td align="left" valign="top">7 (50)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Female</td><td align="left" valign="top">7 (50)</td></tr><tr><td align="left" valign="top" colspan="2">Expertise</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Clinical trial research</td><td align="left" valign="top">1 (7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>AI and informatics</td><td align="left" valign="top">2 (14)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Clinical ophthalmology</td><td align="left" valign="top">3 (21)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Bioethics (clinical AI ethics)</td><td align="left" valign="top">3 (21)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Health policy</td><td align="left" valign="top">2 (14)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Patient advocacy</td><td align="left" valign="top">2 (14)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Industry AI</td><td align="left" valign="top">1 (7)</td></tr></tbody></table></table-wrap></sec><sec id="s2-3"><title>Survey Stages</title><sec id="s2-3-1"><title>Overview</title><p>Two online structured surveys were administered via Qualtrics (Qualtrics, LLC). Participant responses were kept anonymous during the survey rounds to promote candid responses from all participants. Respondents were allowed approximately 1 month to respond, with a reminder email sent 1 to 2 weeks before each deadline to all who accepted the invitation to participate.</p></sec><sec id="s2-3-2"><title>Survey 1</title><p>The first survey was initiated by informing participants that the purpose of the survey was to characterize the ethical issues relevant to the use of AI for clinical screening tools and to inform recommendations for research and policy stakeholders. The first survey elicited participants&#x2019; views on the NIH principles for ethical clinical research [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. A vignette of the first NIH trial of an autonomous AI tool for DR screening was used (<xref ref-type="other" rid="box1">Textbox 1</xref>). The tool served as an exemplar case for expert panelists to consider in the first-round survey, given that it could ground the discussion in a concrete example of autonomous AI for a clinical screening application that had already undergone FDA review. Open-ended questions were used to capture participants&#x2019; views on each of the NIH principles in relation to the clinical trial vignette and to identify conflicts, if any, that potentially arose between value considerations (<xref ref-type="table" rid="table2">Table 2</xref>). The research team analyzed the responses for themes grounded in the NIH principles. We selected representative statements from the responses that addressed potential conflicts or tensions between the ethical principles, such as the tension between scientific benefit and cost-effectiveness. These statements were then used to develop the second survey.</p><boxed-text id="box1"><title> First survey vignette.</title><p>In 2018, the FDA approved the first autonomous AI software that interprets retinal images taken with a non-mydriatic fundus camera, providing an immediate result for diabetic retinopathy (DR) screening at the point of care (POC) for adults with diabetes. The PIs of the trial were the first to implement this technology in pediatrics, demonstrating safety, effectiveness, and equity, and cost-savings to the patient. They also found that minority youth, those with lower household income, and those with Medicaid insurance were less likely to undergo recommended screening, yet were more likely to have DR. The current trial hypothesizes that implementing POC autonomous AI in the diabetes care setting will increase DR screening rates in youth with diabetes, mitigate disparities in access to screening, and be cost-effective to the health care system. A randomized control trial conducted at two clinic sites is meant to determine 1) if autonomous AI increases screening compared to an eye-care professional (ECP), and 2) if those who screen positive by AI are more likely to go for follow-up at the ECP. There is also a prospective observational trial of this AI screening tool to determine if it mitigates disparities in screening and improves the proportion of at-risk, minority, and low-income youth who go for follow-up if their AI screen is positive. If the AI tool is shown to increase screening rates while mitigating disparities in access to care, it has the potential to reshape screening methods now and in the future.</p></boxed-text><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Survey 1 questions and response trends with example quotes.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Question</td><td align="left" valign="bottom">Trends in responses</td><td align="left" valign="bottom">Exemplar quotations</td></tr></thead><tbody><tr><td align="left" valign="top">Question 1: clinical value<list list-type="bullet"><list-item><p>Do you think there are challenges to identifying and assessing clinical value in this context? Please identify.</p></list-item><list-item><p>While AI can improve access to screening for marginalized populations, it may not provide screening comparable to having access to an ophthalmologist. How, if at all, do you think this issue impacts assessment of the clinical value of the tool?</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Several participants stressed the importance of determining the performance of the AI tool relative to screening by an ophthalmologist to clarify if potential gains provided by increasing screening access outweighed potential reduction in diagnostic accuracy.</p></list-item><list-item><p>A few thought that AI tool performance did not need to be comparable to an ophthalmologist to have clinical value.</p></list-item><list-item><p>Several highlighted the issue of the need for clinical follow-up and moral imperative to ensure that those flagged as positive in screening sought appropriate care afterward.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>&#x201C;Clinical value of the tool is not impacted as long as it is clearly shared by the PCP<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> to the patient that this is not comparable screening, but rather a preliminary assessment.&#x201D;</p></list-item><list-item><p>&#x201C;Look at algorithmic performance etc. in comparison with manual workflows...&#x201D;</p></list-item><list-item><p>&#x201C;Value arises from what we do in response to the AI tools output...&#x201D;</p></list-item><list-item><p>&#x201C;I think there is a debatable question about whether it&#x2019;s OK to just sit back and watch how many of the kids who screen positive on AI actually follow up with an ophthalmologist, or whether because this is research, the team has an obligation to monitor who gets appointments on their own and then for those who don&#x2019;t do that within a reasonable period of time, to intervene and make appointments for them.&#x201D;</p></list-item></list></td></tr><tr><td align="left" valign="top">Question 2: social value<break/>Social value is the idea that the research question should contribute to scientific understanding of health or improve our ways of preventing, treating, or caring for people with a given disease to justify exposing participants to the risk and burden of research.<list list-type="bullet"><list-item><p>What do you consider to be the social value of the AI tool in the vignette?</p></list-item><list-item><p>What, if any, approaches do you think can be used to evaluate or address outcomes for that social value?</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Most participants broadly agreed that the social value of the tool was its potential to improve the likelihood that people receive a timely and accurate diagnosis with the potential to narrow disparities for underserved patients.</p></list-item><list-item><p>Some suggestions of approaches were the creation of a data registry for participating sites; sharing of information on AI screening at sites; and patient or community surveys and/or focus groups.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>&#x201C;That it could increase diagnosis of DR<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup> and reduce disparities in DR. I think that quantifying differences in diagnosis across social groups could address the social value of the intervention.&#x201D;</p></list-item><list-item><p>&#x201C;The AI tool per se has zero social value. The social value is created from increasing screening rates while mitigating disparities in access to care.&#x201D;</p></list-item></list></td></tr><tr><td align="left" valign="top">Question 3: potential conflicts<break/>Describe conflicts, if any, that you think could arise between clinical, social, and/or organizational values during development and/or evaluation of a diabetic retinopathy AI screening tool (eg, cost-effectiveness vs. improving care or access)?<list list-type="bullet"><list-item><p>What, if any, are ways to address these conflicts?</p></list-item><list-item><p>How would you ethically test cost-benefit of AI implementation?</p></list-item><list-item><p>Cost can influence test parameters (such as accepting a lower sensitivity &#x0026; specificity for a lower cost). - what, if any, impact do you think this has on ethical considerations?</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Many participants identified potential conflicts thought not to be unique to the use of AI tools, eg, tension between high-sensitivity tests and cost burden.</p></list-item><list-item><p>Several suggested conflicts specific to AI use, including issues of patient and medical professional trust in the use of AI.</p></list-item><list-item><p>Transparency with patients and physicians was thought to be important if cost considerations dictate lower test sensitivity over diagnosis.</p></list-item><list-item><p>Expressed concern over reinforcing &#x201C;tiers of care,&#x201D; with higher access to in-person testing for better-resourced groups and less-resourced groups having only AI screening.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p><italic>&#x201C;...organizations should analyze all streams of costs and benefits and have a multidisciplinary body that assesses prospective AI tool deployments, following criteria that have been agreed up on in advance, which [let&#x2019;s be honest] will likely include financial impact on the organization. b. I don&#x2019;t see distinctive issues in performing cost-benefit or cost-effectiveness analyses for AI, compared to other clinical interventions... c. Organizations need to have total transparency with the physicians who use their tools, and potentially also with patients (depending on the tool), about what is known about model sensitivity and specificity...&#x201D;</italic></p></list-item><list-item><p><italic>&#x201C;The main conflict here is that these efforts need to be sustainable -- and the most under-resourced areas are least likely to be able to afford additional screening, follow-up, and treatment.&#x201D;</italic></p></list-item></list></td></tr><tr><td align="left" valign="top">Question 4: risk-benefit considerations<list list-type="bullet"><list-item><p>In the context of the vignette, what challenges do you see in establishing an appropriate risk-benefit ratio when the risks of AI, or introducing AI into workflows, are not yet fully known?</p></list-item><list-item><p>What concerns come up when applying risk-benefit considerations for this AI tool to different populations? What, if any, are ways to address these challenges?</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Several participants raised the issue that there are multiple unknowns regarding introducing AI into workflows, making a risk-benefit analysis difficult.</p></list-item><list-item><p>Multiple experts mentioned the risks of automation bias and algorithmic bias where risks may vary across different groups and individuals.</p></list-item><list-item><p>Several highlighted difficulties of collecting data on follow-up visits, rates of DR development, acceptance of AI screening etc across different health care systems because of data compatibility and extraction.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p><italic>&#x201C;One of the major reasons that clinical studies are so important is because the risks are not yet known. There are risks of automation bias and automation neglect. The importance of the trial protocol having clear specifications around how the outputs are used by the treating team is important to parsing out the effect of the model outputs specifically vs a general effect of being in a research trial [the greatest cause of bias in observational trials] The risks are also differentially distributed across populations, which we know well at this point [eg, algorithmic bias]. There is no consideration of bias in current regulatory schemes nor do most oversight frameworks consider the effects of fairness beyond the technical performance of the system alone, which is insufficient for addressing fairness in a clinical research context.&#x201D;</italic></p></list-item></list></td></tr><tr><td align="left" valign="top">Question 5: equipoise<list list-type="bullet"><list-item><p>Establishing equipoise can be challenging for AI medical tools because trials have to consider not just if the AI is better than current approaches but also the context into which the AI is to be applied. For an AI tool to screen for diabetic retinopathy, what do you think are important considerations for addressing equipoise?</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Most participants thought the equipoise issue was not that different from that in other medical interventions.</p></list-item><list-item><p>Several mentioned that equipoise should be established, not relative to the AI screening, but to the post-screening steps such as seeking appropriate follow-up testing and care; however, 2 commented that equipoise could be considered based on screening alone.</p></list-item><list-item><p>Several mentioned the need to match the AI training population data to the target setting and patient population.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p><italic>&#x201C;I don&#x2019;t know if I agree that establishing equipoise is challenging for AI tools specifically, I think there&#x2019;s not really that much different to other interventions in medicine where both of those consideration apply.&#x201D;</italic></p></list-item><list-item><p><italic>&#x201C;...perhaps considering clinical outcomes and therapeutic interventions rather than diagnostic screening rates alone.&#x201D;</italic></p></list-item><list-item><p><italic>&#x201C;This is important, but not the issue with screening. It seems problematic to see diagnosis of treatable illnesses as an issue because folks won&#x2019;t get treatment. That seems a separate issue. I think problematic to consider this given it could lead to not wanting to screen among uninsured folks for example.&#x201D;</italic></p></list-item></list></td></tr><tr><td align="left" valign="top">Question 6: access to therapeutic interventions<list list-type="bullet"><list-item><p>AI screening of populations who have difficulty accessing health care may not improve their access to therapeutic interventions, even if it does improve diagnostic screening. How do you think this impacts ethical issues at play in the vignette?</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>There were divided opinions over issues of AI that increased rates of diagnosis but unaccompanied by appropriate follow-up and therapeutic intervention.</p></list-item><list-item><p>Some expressed concern over distribution of resources and investment if patients were not ultimately helped.</p></list-item><list-item><p>Others still saw value in improving diagnostic screening even if follow-ups were not impacted.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p><italic>&#x201C;We will never know disparities in treatment exist if we don&#x2019;t diagnose as many people as possible -- as long as it is something that is meaningful to diagnose.&#x201D;</italic></p></list-item><list-item><p><italic>&#x201C;This is an uncertainty that should be acknowledged. We can draw from lessons learned from other screening or diagnostic programs. In melanoma for example, there is a data to show increased rates of diagnosis does not improve outcomes in the long-term - therefore the impact on resources, cost, patient distress are all consequences of an intervention that may not be helpful.&#x201D;</italic></p></list-item><list-item><p><italic>&#x201C;It shows how technology is always only one piece of a larger problem. The ethical issue then is insufficient scientific practices contribute to research waste...&#x201D;</italic></p></list-item></list></td></tr><tr><td align="left" valign="top">Question 7: potential impact of existing health inequities<list list-type="bullet"><list-item><p>What concerns, if any, would you have for the above scenario regarding how existing health inequities might impact ethical application of the AI tool?</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>A majority of the participants had concerns regarding the impact of existing health inequities on the AI tool application.</p></list-item><list-item><p>Opinions were divided over whether this concern was different relative to non-AI health care interventions.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>&#x201C;This is not much different to current standard in health care and research. The best approach is to consider inclusivity in research design to encourage typically under-represented groups to participate in the trial. If certain groups are less likely to go for screening, you still won&#x2019;t be reaching them just because you added an AI tool to a place where they are already less likely to go. And if you don&#x2019;t change the human aspects of healthcare (eg, anti-racism, inclusive care) then it doesn&#x2019;t matter how good the technology actually is, it won&#x2019;t improve things.&#x201D;</p></list-item><list-item><p>&#x201C;Access is a major problem for implementation - will the tool, if built upon research in this population, make its way back for use by this population?&#x201D;</p></list-item></list></td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>PCP: primary care provider.</p></fn><fn id="table2fn2"><p><sup>b</sup>DR: diabetic retinopathy.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-3-3"><title>Survey 2</title><p>The purpose of the second survey was to further develop an understanding of the issues relevant to AI clinical trials and the 7 principles for ethical research that the participants thought should be prioritized. Participants were asked to rate the representative statements according to the level of agreement with the statement on a Likert scale of 1 to 5 (1 being least and 5 being most; <xref ref-type="other" rid="box2">Textbox 2</xref>). In accordance with Delphi methodology, statements of consensus (at least 80% rating agreement) and moderate agreement (60%&#x2010;80% rating agreement) were identified, along with certain particularly divisive statements (with no more than one neutral rating and otherwise evenly split across the agree and disagree ratings). Responses were descriptively analyzed by the research team, and an initial summary of consensus recommendations was then drafted for the final round. We used Delphi standard cutoffs regarding agreement [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref28">28</xref>].</p><boxed-text id="box2"><title> Survey 2 statements of consensus, moderate agreement, and divisive statements.</title><p>Strong agreement</p><list list-type="bullet"><list-item><p>To evaluate the clinical value of the diabetic retinopathy AI tool, there first needs to be a comparison to the quality of care that would be received when an ophthalmologist does the task.</p></list-item><list-item><p>The social value of an AI tool should be measured by examining the relative gains or losses between different populations, such as the number of people getting diagnosed and people actually getting treatment.</p></list-item></list><p>Risk-benefit ratio</p><list list-type="bullet"><list-item><p>The clinical trials themselves establish the risks involved, and it will be important, as part of that, to establish how these risks are unequally distributed across different populations.</p></list-item><list-item><p>For the evaluation of clinical AI interventions, it is particularly important to identify and share information (eg, characteristics) regarding the target population and research population.</p></list-item><list-item><p>A key problem in clinical evaluation of AI tools is being able to generalize that the AI will be effective in different sites and different populations.</p></list-item><list-item><p>In evaluating clinical AI, it is particularly important to ensure that there is discussion of how bias sensitivity-specificity is set for majority versus minority populations.</p></list-item><list-item><p>As existing inequities will likely impact any potential benefits of the AI tool for specific populations, there needs to be initial investment in understanding those inequities before approving use of AI tools in different populations.</p></list-item><list-item><p>For consent purposes, it is important that patients understand that their data are being used for clinical trials as well as potentially for other uses, such as developing pharmaceuticals.</p></list-item><list-item><p>There need to be more clearly delineated avenues for patients to contribute their perspectives to the process of evaluating clinical AI tools.</p></list-item></list><p>Moderate agreement</p><list list-type="bullet"><list-item><p>There is a need for a new ethical framework that does more to anticipate downstream implications or social value of the AI tool being clinically evaluated.</p></list-item><list-item><p>Social value for an AI tool cannot be evaluated as a stand-alone issue because the social value comes from the related systemic support factors, such as having structural factors in place to increase screening, diagnosis, and treatment overall.</p></list-item></list><p>Divisive statements</p><list list-type="bullet"><list-item><p>Evaluating a clinical AI intervention is not significantly different from evaluating any other type of clinical intervention.</p></list-item><list-item><p>Evaluating clinical AI is different from evaluating other types of clinical interventions because the outcome of an AI intervention may not be transferable to different sites or populations.</p></list-item><list-item><p>Evaluating clinical AI is different from evaluating other clinical interventions because the outcome of an AI intervention is often more dependent on systemic factors, such as access to additional care afterward.</p></list-item><list-item><p>Evaluation of the cost-effectiveness of an AI tool needs to be broken down according to different populations.</p></list-item></list></boxed-text></sec><sec id="s2-3-4"><title>Final Virtual Meeting</title><p>The final round of the Delphi study took place as a virtual meeting using Zoom (Zoom Communications, Inc) for group discussion. There was a structured discussion in which the panelists reviewed the results of survey 2, focusing the discussion on the statements for which there was consensus, as well as the statements that were most polarized between strongly agree and strongly disagree in the responses. The discussion also allowed panelists to clarify some of the reasoning behind the statements, as well as to endorse revisions to the statements as a group. In this way, the meeting served to finalize the recommendations by the expert panel.</p></sec></sec><sec id="s2-4"><title>Ethical Considerations</title><p>The protocol for this study was approved by Stanford&#x2019;s Institutional Review Board (Protocol 69318). Informed consent was obtained, with a waiver of documentation approved by the institutional review board. Data from surveys 1 and 2 were anonymized. No compensation was used.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Survey 1</title><p>Of the 14 experts who agreed to participate in the panel, 12 (85.7%) responded to survey 1. The questions and exemplar quotes from survey 1 respondents are presented in <xref ref-type="table" rid="table2">Table 2</xref>.</p><p>Participant responses regarding the ethical principle of &#x201C;Clinical and Social Value&#x201D; indicated several key tensions in applying this principle to the evaluation of an autonomous AI tool. For example, several participants identified the need for ensuring clinical follow-up as a potential challenge to substantiating the clinical value of an AI tool. Most participants broadly saw potential social value, that is, that research should contribute to scientific understanding of health or improve ways of preventing, treating, or caring for people with a given disease to justify participants&#x2019; exposure to the risk and burden of research. The social value was reported as the potential to provide a more timely and accurate diagnosis for a disease with treatment available for mitigating further disease progression and to narrow the gap in access to screening for underserved populations. Participants also provided several suggestions for how to evaluate or address social value further, which included community focus groups and a data registry for participating sites.</p><p>Most participants raised cost considerations as potentially conflicting with the clinical or social value of the AI tool. At the same time, most participants felt that the cost-benefit analysis of autonomous AI is no different than the assessment of non-AI health care interventions. However, some responses reported that a concern specific to AI tools in health care was the potential impact of patient, public, and providers&#x2019; trust in an AI tool on the clinical impact of the tool.</p><p>Regarding the principle of &#x201C;Favorable Risk-Benefit Ratio,&#x201D; the risk-benefit considerations of autonomous AI were seen to be multifactorial with variable unknowns regarding the introduction of AI into workflows and barriers to data extraction and collection, making risk-benefit analysis difficult. Participants also broadly reported that the consideration of clinical equipoise for AI-based clinical trials was similar to those of non-AI clinical trials.</p><p>The issue of health disparities was seen in survey responses regarding several of the principles for ethical research, such as &#x201C;Clinical and Social Value,&#x201D; &#x201C;Fair Subject Selection,&#x201D; and &#x201C;Respect for Enrolled Subjects.&#x201D; The majority of participants had concerns regarding the impact of existing disparities in access to care on the AI tool application development and bias. Participant responses varied on the relevance of access to diagnostic screening without sufficiently addressing barriers to access to therapeutic care. Most participants emphasized that this was not a new issue in clinical research. A few participants posited autonomous AI clinical trials as potential research waste, where resources could be allocated elsewhere to improve care access. Some participants saw access to screening as beneficial, with one stating that screened patients would still have the knowledge of the presence of a health irregularity and may access care down the road.</p></sec><sec id="s3-2"><title>Survey 2</title><p>Of the 14 experts who agreed to participate in the panel, 10 (71.4%) responded to survey 2. Two experts who missed the survey deadline emailed written comments on the survey questions that were incorporated into the survey 2 findings. <xref ref-type="other" rid="box2">Textbox 2</xref> summarizes statements of consensus (at least 80% rating agreement), moderate agreement (60%&#x2010;80% rating agreement), and divisive statements (with no more than one neutral rating and otherwise panelists evenly split across agree and disagree ratings).</p></sec><sec id="s3-3"><title>Final Round Virtual Meeting</title><p>The final round meeting had both synchronous and asynchronous participation by 13 (92.9%) of 14 participants. Panelists identified reducing bias, improving access to care, ensuring downstream access to recommended treatments, promoting transparency regarding the representativeness of the training data, determining prespecified outcomes with respect to the reference standard, having clear pretrial discussion of cost or cost savings, and engaging relevant stakeholders (eg, clinicians and patients) as central priorities. A broader theme derived from the discussion was a tension that although many ethical concerns in AI trials resemble longstanding issues in clinical research, autonomous AI trials may provide an opportunity to address these issues more systematically. The following subsections summarize the combined results.</p></sec><sec id="s3-4"><title>Opportunities to Advance Ethical Research Principles</title><p>Participating panelists emphasized that AI tools are distinct from non-AI medical interventions, yet the issues faced in conducting clinical trials are similar. Panelists who expressed the position that AI tools did not present significant differences for clinical trials tended to point out that priority ethical issues for AI, such as bias or informed consent, have also, historically, presented problems for traditional medical interventions. Given the current investment in AI devices, panelists saw there being a potentially useful moment for AI developers to reflect on bias and avoid the development of medical devices that would unintentionally amplify those inequalities. Some panelists pointed toward ways that they saw AI presenting distinct ethical challenges, including clinicians&#x2019; limited understanding of potential AI tool limitations and the need to ensure context-specific usability within diverse clinical workflows. Panelists agreed on the existence, but not the magnitude, of these differences and on whether such differences warrant adjustments to the ethical framework for clinical trials. Regardless of the distinctiveness of AI evaluation, expert panelists agreed that efforts to improve clinical trials of AI could also provide an important opportunity to address longstanding ethical issues of bias, informed consent, and access to care.</p></sec><sec id="s3-5"><title>Clinical Value</title><p>The expert panel highlighted the need for trials to prespecify outcomes with respect to the reference standard and to describe any challenges in comparison, such as situations in which the current professional medical examination screens for multiple conditions simultaneously, while the AI tool screens for only one condition. Panelists frequently used the term &#x201C;trial design&#x201D; broadly to include aspects such as randomization strategy, comparator arm selection, and primary endpoint definitions.</p><p>Several of the panelists raised questions of how the autonomous AI tool for DR screening should be compared to screening by an ophthalmologist. Panelists were interested in this comparison as part of considering the broader issue of whether reliance on AI for DR screening as a cost-effective tool in low-resource settings could create or reinforce a lower tier of care. In the vignette considered by the panel, the AI tool had been evaluated against a level 1 reference standard based on clinical outcome, indicating that the AI for DR screening tool was more accurate than a human clinician at screening for DR, as this was raised by one of the panelists. Nevertheless, panelists agreed that trial reports should clearly explain the chosen reference standard and how it relates to the clinical care patients would otherwise receive.</p></sec><sec id="s3-6"><title>Bias and Stakeholder Engagement</title><p>The panelists endorsed several approaches that are meant to address the use of AI across different contexts and for populations that may not be sufficiently represented in training datasets. The clinical trial of an AI tool for DR screening demonstrated that, when placed at point-of-care sites, the tool could improve screening rates among populations historically lacking access, and its performance did not differ across minoritized subgroup populations. However, panelists raised concerns more broadly that a lack of sufficient representation of minoritized populations in training data can lead to systematic bias that undermines the potential benefits of increased screening. There was general agreement among the panelists of the importance of transparency and sharing of information regarding the demographics of training datasets.</p><p>The impact of context and integration into workflow on the effectiveness and accuracy of AI tools was highlighted as a particularly challenging issue for evaluation. Panelists supported practices of stakeholder engagement for the development and implementation of AI tools in health care. Panelists noted, for example, engaging clinicians regarding practical aspects of how the AI tool might fit into their workflow could also identify the way that use of the tool in those different contexts could support or create challenges for effective use. One concern raised was that a tool that only evaluates for one condition may not be as useful as a tool that can evaluate several potential eye conditions.</p><p>Several panelists discussed how community and/or patient engagement could also be used to help identify factors that impact the usefulness of AI tools for specific purposes and that these kinds of engagement could help identify systemic factors that affect bias and fairness. Panelists agreed that the specific stakeholders needing to be engaged might vary according to the type of tool and purpose. Panelists noted that stakeholder engagement has become more common in recent years for clinical research, and they recognized the need to include such engagement with AI to support these kinds of efforts regularly as part of the process of development and implementation.</p></sec><sec id="s3-7"><title>Cost-Effectiveness</title><p>Several panelists expressed concern that, generally speaking, AI could make it easier to implement different tiers of care, where the price point of an AI tool could be used to lower costs as well as quality of care for lower-resourced patients. While that was not as likely the case with the specific AI tool for DR screening, given studies found that the tool performed better than clinicians on average, panelists wanted to note this issue for the field in general. At the same time, panelists acknowledged that such concerns regarding cost are not necessarily unique to AI tools within health care. Several panelists suggested that cost savings with an AI tool could present an opportunity within a health care institution or system in ways that benefited underserved patients, such as increasing support for additional access to treatment.</p></sec><sec id="s3-8"><title>Downstream Implications of the AI Tool</title><p>Panelists gave substantial attention to the downstream implications of the AI tool for DR screening and whether evaluation of the tool should account for access to treatment options after a positive screen. Several panelists expressed ethical concerns over an AI tool that might increase the number of people identified as potentially having DR but then leave those people without a way to address their condition. Participants noted that this concern of a diagnosis without access to treatment is certainly not limited to AI tools and that clinical trials are generally not expected to resolve system-level gaps in access. However, the majority of panelists agreed that such downstream implications should be considered when evaluating when and how to use an autonomous AI tool. A few panelists also raised a potential benefit that having an AI screening tool implemented might free up clinician time for treating rather than screening. They noted, however, that whether this benefit would be realized would still depend upon the resources and structure of the health system in which the screening takes place.</p><p>The Delphi panel concluded that to apply the 7 ethical principles of research to autonomous AI clinical trials, it is particularly important to incorporate evaluation of potential downstream implications, such as access. Panelists acknowledged that a number of the concerns that they prioritized for ethical clinical research with AI could be applied to conventional medical interventions as well, such as addressing the risk of bias and incorporating engagement with patients and other stakeholders. At the same time, however, the expert panelists overall agreed that addressing the framework for research ethics of autonomous AI clinical trials could also be seen as a greater opportunity to advocate for structural changes that promoted equity more broadly in clinical research and health care.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This Delphi study identified several areas of expert consensus regarding how the 7 NIH ethical research principles should be applied to clinical trials of autonomous AI. Consensus centered on practical features of trial design and reporting, including documentation of training and validation data, prespecified outcomes and reference standards, assessment of performance across populations and settings, engagement with relevant stakeholders, and consideration of cost, access, and follow-up care. Although panelists differed on the extent to which they viewed autonomous AI as distinct from conventional medical interventions, they agreed that many concerns raised by autonomous AI tools reflected longstanding challenges in clinical research. At the same time, panelists viewed the current development of autonomous AI tools as an opportunity to address such concerns more systematically within clinical trial design and evaluation. Overall, these findings suggest that applying established research ethics principles to autonomous AI clinical trials requires evaluating tool performance, while also considering downstream effects on patients, clinicians, and health systems.</p><p>These findings extend existing applications of the NIH research ethics principles by specifying how clinical value, scientific validity, and equity should be considered in autonomous AI clinical trials. The panel&#x2019;s emphasis on downstream implications aligns with prior work emphasizing that ethical evaluation of medical AI depends not only on model performance but also on implementation context and real-world deployment [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]. This focus is also consistent with qualitative evidence showing that clinical AI implementation is shaped by interdependent factors across multiple stakeholder groups [<xref ref-type="bibr" rid="ref30">30</xref>] and that developers can recognize potential AI-related harms to patients, groups, and health systems though they vary in their level of perceived responsibility in mitigating harms [<xref ref-type="bibr" rid="ref31">31</xref>]. The panel&#x2019;s prioritization of bias and representativeness is further supported by evidence that addressing AI bias requires both technical and social approaches [<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref32">32</xref>], that most FDA-evaluated AI tools do not report demographic characteristics of training data [<xref ref-type="bibr" rid="ref33">33</xref>], and that most AI clinical trials internationally are single-center studies [<xref ref-type="bibr" rid="ref34">34</xref>]. Qualitative work on AI health datasets also supports the importance of documenting data representativeness and intended use in terms of societal impact [<xref ref-type="bibr" rid="ref35">35</xref>]. Taken together, our Delphi study and the existing literature support incorporating attention to representativeness, subgroup performance, downstream access, and lifecycle costs into the ethical evaluation of autonomous AI clinical trials, particularly for tools intended for low-resource settings.</p><p>Our earlier qualitative study of clinical trials involving AI for DR screening identified a range of ethical challenges that were not fully addressed by existing research ethics frameworks, including issues related to social value, scientific validity, and informed consent [<xref ref-type="bibr" rid="ref21">21</xref>]. However, the participants in that study were restricted to investigators involved in AI clinical trials for the specific purpose of screening for DR. By engaging a broader, multidisciplinary panel of experts, including experts in health policy and patient advocacy, panelists were able to reach consensus on several key overarching issues affecting the application of ethical research principles to AI clinical trials. The study resulted in the development of 9 key actionable recommendations (<xref ref-type="table" rid="table3">Table 3</xref>). Implementation of these actions would help regulators and investigators promote equitable potential benefits and reduce potential unintended consequences of autonomous AI clinical trials.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Application of ethical considerations in real-world clinical AI trial settings.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Action items</td><td align="left" valign="bottom">Description</td></tr></thead><tbody><tr><td align="left" valign="top">Increase transparency of AI training and validation data</td><td align="left" valign="top">Trial protocols should describe how the training data compare with the intended trial population and identify any representativeness gaps before enrollment begins.</td></tr><tr><td align="left" valign="top">Assess bias before deployment</td><td align="left" valign="top">The statistical analysis plan should include prespecified subgroup analyses and define what level of performance difference would require modification, monitoring, or nondeployment.</td></tr><tr><td align="left" valign="top">Evaluate performance across clinical settings</td><td align="left" valign="top">Trials should include sites that reflect the settings where the AI tool is likely to be used and assess whether performance changes across care environments.</td></tr><tr><td align="left" valign="top">Address health inequities before implementation</td><td align="left" valign="top">Investigators should identify structural barriers that may affect AI performance or patient outcomes and describe mitigation steps before widespread use.</td></tr><tr><td align="left" valign="top">Incorporate stakeholder engagement</td><td align="left" valign="top">Trial development should document how input from affected patients, clinicians, or community representatives shaped study design and implementation.</td></tr><tr><td align="left" valign="top">Strengthen informed consent practices</td><td align="left" valign="top">Consent materials should explain the role of the AI tool in care decisions and describe whether patient data may be used beyond the immediate trial.</td></tr><tr><td align="left" valign="top">Compare AI tools with existing standards of care</td><td align="left" valign="top">AI tools should be evaluated against current clinical practice, including effects on diagnostic performance and patient-relevant outcomes.</td></tr><tr><td align="left" valign="top">Evaluate downstream care access</td><td align="left" valign="top">Trials should assess whether patients can obtain appropriate follow-up care after AI-generated recommendations, rather than measuring accuracy alone.</td></tr><tr><td align="left" valign="top">Assess cost and access implications</td><td align="left" valign="top">Trial designs should take into account whether AI implementation could shift costs to patients, clinics, or under-resourced health systems.</td></tr></tbody></table></table-wrap></sec><sec id="s4-2"><title>Limitations</title><p>The study has several limitations. While the number of experts on the Delphi panel was within the recommended range for Delphi practices and was highly multidisciplinary, the panel may not represent the full range of perspectives. In addition, all expert panelists were based in the United States, which could limit the international applicability of the recommendations in terms of familiarity with other practice and regulatory contexts. As a Delphi study, the approach is based on the idea that structured discussion and review of the research questions by relevant experts yields practical guidance. Additionally, the regulatory and technological landscape for autonomous AI is rapidly evolving, and therefore, these recommendations may require ongoing refinement as new tools emerge. This study also did not directly address environmental sustainability associated with AI development and monitoring, which could be an area of future consideration.</p></sec><sec id="s4-3"><title>Conclusions</title><p>Autonomous AI clinical trials raise ethical considerations that extend beyond technical accuracy, particularly when tools are intended for use across diverse populations and health care settings. The recommendations developed through this Delphi process suggest that ethical evaluation of these trials should account for the provenance and representativeness of training and validation data, the possibility of biased performance, stakeholder perspectives, access to follow-up care, and cost-related effects. As autonomous AI tools continue to be tested in clinical research, these considerations can help shape regulatory and policy expectations for the ethical development and evaluation of autonomous AI clinical trials.</p></sec></sec></body><back><ack><p>We extend our sincere gratitude to the Delphi expert panelists. Their invaluable contributions, dedicated time, and willingness to share their insights have been fundamental to the success of this research.</p></ack><notes><sec><title>Funding</title><p>This study was funded by the National Eye Institute (R01EY033233-01).</p></sec><sec><title>Data Availability</title><p>Participant data will not be shared. Survey instruments of this study can be shared upon request to the corresponding author.</p></sec></notes><fn-group><fn fn-type="con"><p>RMW, DC, NM-M, AY, and MA contributed to study conceptualization and funding acquisition. AY, AAN, NM-M, DBL, RMW, and DC contributed to data curation, methodology, and formal analysis. AAN and NM-M contributed to the original manuscript draft. All coauthors contributed to manuscript editing and review.</p></fn><fn fn-type="conflict"><p>DBL reported holding shares in Bunker Hill Health Shareholder outside the submitted work and receiving research support from Siemens Healthineers and the Gordon and Betty Moore Foundation outside the submitted work. MA reported holding roles as director and consultant with Digital Diagnostics Inc; chairing the Healthcare AI Coalition Foundational Principles of AI Collaborative Community for Ophthalmic Imaging; serving as committee member of the American Academy of Ophthalmology AI Committee, AI Workgroup Digital Medicine Payment Advisory Group, and the Collaborative Community for Ophthalmic Imaging outside the submitted work. RMW reported grants from Novo Nordisk as primary investigator for a clinical research site outside the submitted work. No other disclosures were reported.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">DR</term><def><p> diabetic retinopathy</p></def></def-item><def-item><term id="abb2">FDA</term><def><p>US Food and Drug Administration</p></def></def-item><def-item><term id="abb3">NIH </term><def><p>National Institutes of Health</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="web"><article-title>Artificial intelligence-enabled medical devices</article-title><source>US Food and Drug Administration</source><access-date>2025-09-10</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.fda.gov/medical-devices/software-medical-device-samd/artificial-intelligence-and-machine-learning-aiml-enabled-medical-devices">https://www.fda.gov/medical-devices/software-medical-device-samd/artificial-intelligence-and-machine-learning-aiml-enabled-medical-devices</ext-link></comment></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McCradden</surname><given-names>MD</given-names> </name><name name-style="western"><surname>Anderson</surname><given-names>JA</given-names> </name><name name-style="western"><surname>A Stephenson</surname><given-names>E</given-names> </name><etal/></person-group><article-title>A research ethics framework for the clinical translation of healthcare machine learning</article-title><source>Am J Bioeth</source><year>2022</year><month>05</month><volume>22</volume><issue>5</issue><fpage>8</fpage><lpage>22</lpage><pub-id pub-id-type="doi">10.1080/15265161.2021.2013977</pub-id><pub-id pub-id-type="medline">35048782</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abr&#x00E0;moff</surname><given-names>MD</given-names> </name><name name-style="western"><surname>Cunningham</surname><given-names>B</given-names> </name><name name-style="western"><surname>Patel</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Foundational considerations for artificial intelligence using ophthalmic images</article-title><source>Ophthalmology</source><year>2022</year><month>02</month><volume>129</volume><issue>2</issue><fpage>e14</fpage><lpage>e32</lpage><pub-id pub-id-type="doi">10.1016/j.ophtha.2021.08.023</pub-id><pub-id pub-id-type="medline">34478784</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abr&#x00E0;moff</surname><given-names>MD</given-names> </name><name name-style="western"><surname>Tarver</surname><given-names>ME</given-names> </name><name name-style="western"><surname>Loyo-Berrios</surname><given-names>N</given-names> </name><collab>Foundational Principles of Ophthalmic Imaging and Algorithmic Interpretation Working Group of the Collaborative Community for Ophthalmic Imaging Foundation, Washington, D.C</collab><etal/></person-group><article-title>Considerations for addressing bias in artificial intelligence for health equity</article-title><source>NPJ Digit Med</source><year>2023</year><month>09</month><day>12</day><volume>6</volume><issue>1</issue><fpage>170</fpage><pub-id pub-id-type="doi">10.1038/s41746-023-00913-9</pub-id><pub-id pub-id-type="medline">37700029</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Grote</surname><given-names>T</given-names> </name></person-group><article-title>Randomised controlled trials in medical AI: ethical considerations</article-title><source>J Med Ethics</source><year>2022</year><month>11</month><volume>48</volume><issue>11</issue><fpage>899</fpage><lpage>906</lpage><pub-id pub-id-type="doi">10.1136/medethics-2020-107166</pub-id><pub-id pub-id-type="medline">33990429</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="book"><person-group person-group-type="editor"><name name-style="western"><surname>Matheny</surname><given-names>M</given-names> </name><name name-style="western"><surname>Israni</surname><given-names>ST</given-names> </name><name name-style="western"><surname>Ahmed</surname><given-names>M</given-names> </name><name name-style="western"><surname>Whicher</surname><given-names>D</given-names> </name></person-group><source>Artificial Intelligence in Health Care</source><year>2019</year><publisher-name>National Academies Press</publisher-name><pub-id pub-id-type="doi">10.17226/27111</pub-id><pub-id pub-id-type="other">978-0-309-70513-4</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abramoff</surname><given-names>MD</given-names> </name><name name-style="western"><surname>Dai</surname><given-names>T</given-names> </name><name name-style="western"><surname>Zou</surname><given-names>J</given-names> </name></person-group><article-title>Scaling adoption of medical AI &#x2014; reimbursement from value-based care and fee-for-service perspectives</article-title><source>NEJM AI</source><year>2024</year><month>04</month><day>12</day><volume>1</volume><issue>5</issue><pub-id pub-id-type="doi">10.1056/AIpc2400083</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Challen</surname><given-names>R</given-names> </name><name name-style="western"><surname>Denny</surname><given-names>J</given-names> </name><name name-style="western"><surname>Pitt</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gompels</surname><given-names>L</given-names> </name><name name-style="western"><surname>Edwards</surname><given-names>T</given-names> </name><name name-style="western"><surname>Tsaneva-Atanasova</surname><given-names>K</given-names> </name></person-group><article-title>Artificial intelligence, bias and clinical safety</article-title><source>BMJ Qual Saf</source><year>2019</year><month>03</month><volume>28</volume><issue>3</issue><fpage>231</fpage><lpage>237</lpage><pub-id pub-id-type="doi">10.1136/bmjqs-2018-008370</pub-id><pub-id pub-id-type="medline">30636200</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Obermeyer</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Powers</surname><given-names>B</given-names> </name><name name-style="western"><surname>Vogeli</surname><given-names>C</given-names> </name><name name-style="western"><surname>Mullainathan</surname><given-names>S</given-names> </name></person-group><article-title>Dissecting racial bias in an algorithm used to manage the health of populations</article-title><source>Science</source><year>2019</year><month>10</month><day>25</day><volume>366</volume><issue>6464</issue><fpage>447</fpage><lpage>453</lpage><pub-id pub-id-type="doi">10.1126/science.aax2342</pub-id><pub-id pub-id-type="medline">31649194</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Obermeyer</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Emanuel</surname><given-names>EJ</given-names> </name></person-group><article-title>Predicting the future - big data, machine learning, and clinical medicine</article-title><source>N Engl J Med</source><year>2016</year><month>09</month><day>29</day><volume>375</volume><issue>13</issue><fpage>1216</fpage><lpage>1219</lpage><pub-id pub-id-type="doi">10.1056/NEJMp1606181</pub-id><pub-id pub-id-type="medline">27682033</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="report"><article-title>Artificial intelligence for health</article-title><year>2024</year><access-date>2026-07-14</access-date><publisher-name>World Health Organization</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://cdn.who.int/media/docs/default-source/digital-health-documents/who_brochure_ai_web.pdf?sfvrsn=aa4f4e3b_3&#x0026;download=true">https://cdn.who.int/media/docs/default-source/digital-health-documents/who_brochure_ai_web.pdf?sfvrsn=aa4f4e3b_3&#x0026;download=true</ext-link></comment></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="report"><article-title>Good machine learning practice for medical device development: guiding principles</article-title><year>2025</year><access-date>2026-07-15</access-date><publisher-name>International Medical Device Regulators Forum</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.imdrf.org/sites/default/files/2025-02/IMDRF_AIML%20WG_GMLP_N88%20Final.pdf">https://www.imdrf.org/sites/default/files/2025-02/IMDRF_AIML WG_GMLP_N88 Final.pdf</ext-link></comment></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="web"><article-title>Predetermined change control plans for machine learning-enabled medical devices: guiding principles</article-title><source>US Food and Drug Administration</source><year>2025</year><access-date>2026-05-29</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.fda.gov/medical-devices/software-medical-device-samd/predetermined-change-control-plans-machine-learning-enabled-medical-devices-guiding-principles">https://www.fda.gov/medical-devices/software-medical-device-samd/predetermined-change-control-plans-machine-learning-enabled-medical-devices-guiding-principles</ext-link></comment></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Emanuel</surname><given-names>EJ</given-names> </name><name name-style="western"><surname>Wendler</surname><given-names>D</given-names> </name><name name-style="western"><surname>Grady</surname><given-names>C</given-names> </name></person-group><article-title>What makes clinical research ethical?</article-title><source>JAMA</source><year>2000</year><volume>283</volume><issue>20</issue><fpage>2701</fpage><lpage>2711</lpage><pub-id pub-id-type="doi">10.1001/jama.283.20.2701</pub-id><pub-id pub-id-type="medline">10819955</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="web"><article-title>Guiding principles ethical research</article-title><source>National Institutes for Health</source><access-date>2025-09-10</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.nih.gov/health-information/nih-clinical-research-trials-you/guiding-principles-ethical-research">https://www.nih.gov/health-information/nih-clinical-research-trials-you/guiding-principles-ethical-research</ext-link></comment></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Char</surname><given-names>D</given-names> </name><name name-style="western"><surname>Abr&#x00E0;moff</surname><given-names>M</given-names> </name><name name-style="western"><surname>Feudtner</surname><given-names>C</given-names> </name></person-group><article-title>A framework to evaluate ethical considerations with ML-HCA applications-valuable, even necessary, but never comprehensive</article-title><source>Am J Bioeth</source><year>2020</year><month>11</month><volume>20</volume><issue>11</issue><fpage>W6</fpage><lpage>W10</lpage><pub-id pub-id-type="doi">10.1080/15265161.2020.1827695</pub-id><pub-id pub-id-type="medline">33103985</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Seyyed-Kalantari</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><name name-style="western"><surname>McDermott</surname><given-names>MBA</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>IY</given-names> </name><name name-style="western"><surname>Ghassemi</surname><given-names>M</given-names> </name></person-group><article-title>Underdiagnosis bias of artificial intelligence algorithms applied to chest radiographs in under-served patient populations</article-title><source>Nat Med</source><year>2021</year><month>12</month><volume>27</volume><issue>12</issue><fpage>2176</fpage><lpage>2182</lpage><pub-id pub-id-type="doi">10.1038/s41591-021-01595-0</pub-id><pub-id pub-id-type="medline">34893776</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kropp</surname><given-names>M</given-names> </name><name name-style="western"><surname>Golubnitschaja</surname><given-names>O</given-names> </name><name name-style="western"><surname>Mazurakova</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Diabetic retinopathy as the leading cause of blindness and early predictor of cascading complications-risks and mitigation</article-title><source>EPMA J</source><year>2023</year><month>03</month><volume>14</volume><issue>1</issue><fpage>21</fpage><lpage>42</lpage><pub-id pub-id-type="doi">10.1007/s13167-023-00314-8</pub-id><pub-id pub-id-type="medline">36866156</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wolf</surname><given-names>RM</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>TYA</given-names> </name><name name-style="western"><surname>Thomas</surname><given-names>C</given-names> </name><etal/></person-group><article-title>The SEE study: safety, efficacy, and equity of implementing autonomous artificial intelligence for diagnosing diabetic retinopathy in youth</article-title><source>Diabetes Care</source><year>2021</year><month>03</month><volume>44</volume><issue>3</issue><fpage>781</fpage><lpage>787</lpage><pub-id pub-id-type="doi">10.2337/dc20-1671</pub-id><pub-id pub-id-type="medline">33479160</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wolf</surname><given-names>RM</given-names> </name><name name-style="western"><surname>Channa</surname><given-names>R</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>TYA</given-names> </name><etal/></person-group><article-title>Autonomous artificial intelligence increases screening and follow-up for diabetic retinopathy in youth: the ACCESS randomized control trial</article-title><source>Nat Commun</source><year>2024</year><month>01</month><day>11</day><volume>15</volume><issue>1</issue><fpage>421</fpage><pub-id pub-id-type="doi">10.1038/s41467-023-44676-z</pub-id><pub-id pub-id-type="medline">38212308</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Youssef</surname><given-names>A</given-names> </name><name name-style="western"><surname>Nichol</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Martinez-Martin</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Ethical considerations in the design and conduct of clinical trials of artificial intelligence</article-title><source>JAMA Netw Open</source><year>2024</year><month>09</month><day>3</day><volume>7</volume><issue>9</issue><fpage>e2432482</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.32482</pub-id><pub-id pub-id-type="medline">39240560</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Helmer-Hirschberg</surname><given-names>O</given-names> </name></person-group><article-title>Analysis of the future: the Delphi method</article-title><source>RAND</source><year>1967</year><access-date>2024-04-29</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.rand.org/pubs/papers/P3558.html">https://www.rand.org/pubs/papers/P3558.html</ext-link></comment></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hohmann</surname><given-names>E</given-names> </name><name name-style="western"><surname>Brand</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Rossi</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Lubowitz</surname><given-names>JH</given-names> </name></person-group><article-title>Expert opinion Is necessary: Delphi panel methodology facilitates a scientific approach to consensus</article-title><source>Arthroscopy</source><year>2018</year><month>02</month><volume>34</volume><issue>2</issue><fpage>349</fpage><lpage>351</lpage><pub-id pub-id-type="doi">10.1016/j.arthro.2017.11.022</pub-id><pub-id pub-id-type="medline">29413182</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Hsu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Sandford</surname><given-names>B</given-names> </name></person-group><article-title>The Delphi technique: making sense of consensus</article-title><source>Practical Assessment, Research &#x0026; Evaluation</source><year>2007</year><access-date>2026-07-15</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://openpublishing.library.umass.edu/pare/article/id/1418/">https://openpublishing.library.umass.edu/pare/article/id/1418/</ext-link></comment></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Witkin</surname><given-names>BR</given-names> </name><name name-style="western"><surname>Altschuld</surname><given-names>JW</given-names> </name></person-group><source>Planning and Conducting Needs Assessments: A Practical Guide</source><year>1995</year><access-date>2026-08-16</access-date><publisher-name>SAGE Publications</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://onlinelibrary.wiley.com/doi/abs/10.1002/hrdq.3920070410">https://onlinelibrary.wiley.com/doi/abs/10.1002/hrdq.3920070410</ext-link></comment></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Niederberger</surname><given-names>M</given-names> </name><name name-style="western"><surname>Spranger</surname><given-names>J</given-names> </name></person-group><article-title>Delphi technique in health sciences: a map</article-title><source>Front Public Health</source><year>2020</year><volume>8</volume><fpage>457</fpage><pub-id pub-id-type="doi">10.3389/fpubh.2020.00457</pub-id><pub-id pub-id-type="medline">33072683</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Boulkedid</surname><given-names>R</given-names> </name><name name-style="western"><surname>Abdoul</surname><given-names>H</given-names> </name><name name-style="western"><surname>Loustau</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sibony</surname><given-names>O</given-names> </name><name name-style="western"><surname>Alberti</surname><given-names>C</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Wright</surname><given-names>JM</given-names> </name></person-group><article-title>Using and reporting the Delphi method for selecting healthcare quality indicators: a systematic review</article-title><source>PLoS ONE</source><year>2011</year><volume>6</volume><issue>6</issue><fpage>e20476</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0020476</pub-id><pub-id pub-id-type="medline">21694759</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nadir</surname><given-names>NA</given-names> </name><name name-style="western"><surname>Hart</surname><given-names>D</given-names> </name><name name-style="western"><surname>Cassara</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Simulation-based remediation in emergency medicine residency training: a consensus study</article-title><source>West J Emerg Med</source><year>2019</year><month>01</month><volume>20</volume><issue>1</issue><fpage>145</fpage><lpage>156</lpage><pub-id pub-id-type="doi">10.5811/westjem.2018.10.39781</pub-id><pub-id pub-id-type="medline">30643618</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>London</surname><given-names>AJ</given-names> </name></person-group><article-title>Artificial intelligence in medicine: overcoming or recapitulating structural challenges to improving patient care?</article-title><source>Cell Rep Med</source><year>2022</year><month>05</month><day>17</day><volume>3</volume><issue>5</issue><fpage>100622</fpage><pub-id pub-id-type="doi">10.1016/j.xcrm.2022.100622</pub-id><pub-id pub-id-type="medline">35584620</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hogg</surname><given-names>HDJ</given-names> </name><name name-style="western"><surname>Al-Zubaidy</surname><given-names>M</given-names> </name><collab>Technology Enhanced Macular Services Study Reference Group</collab><etal/></person-group><article-title>Stakeholder perspectives of clinical artificial intelligence implementation: systematic review of qualitative evidence</article-title><source>J Med Internet Res</source><year>2023</year><month>01</month><day>10</day><volume>25</volume><fpage>e39742</fpage><pub-id pub-id-type="doi">10.2196/39742</pub-id><pub-id pub-id-type="medline">36626192</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nichol</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Sankar</surname><given-names>PL</given-names> </name><name name-style="western"><surname>Halley</surname><given-names>MC</given-names> </name><name name-style="western"><surname>Federico</surname><given-names>CA</given-names> </name><name name-style="western"><surname>Cho</surname><given-names>MK</given-names> </name></person-group><article-title>Developer perspectives on potential harms of machine learning predictive analytics in health care: qualitative analysis</article-title><source>J Med Internet Res</source><year>2023</year><month>11</month><day>16</day><volume>25</volume><fpage>e47609</fpage><pub-id pub-id-type="doi">10.2196/47609</pub-id><pub-id pub-id-type="medline">37971798</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="book"><person-group person-group-type="editor"><name name-style="western"><surname>Bibbins-Domingo</surname><given-names>K</given-names> </name><name name-style="western"><surname>Helman</surname><given-names>A</given-names> </name></person-group><article-title>Barriers to representation of underrepresented and excluded populations in clinical research</article-title><source>Improving Representation in Clinical Trials and Research: Building Research Equity for Women and Underrepresented Groups</source><year>2022</year><access-date>2026-07-14</access-date><publisher-name>National Academies Press</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/books/NBK584407/">https://www.ncbi.nlm.nih.gov/books/NBK584407/</ext-link></comment></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>E</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>K</given-names> </name><name name-style="western"><surname>Daneshjou</surname><given-names>R</given-names> </name><name name-style="western"><surname>Ouyang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Ho</surname><given-names>DE</given-names> </name><name name-style="western"><surname>Zou</surname><given-names>J</given-names> </name></person-group><article-title>How medical AI devices are evaluated: limitations and recommendations from an analysis of FDA approvals</article-title><source>Nat Med</source><year>2021</year><month>04</month><volume>27</volume><issue>4</issue><fpage>582</fpage><lpage>584</lpage><pub-id pub-id-type="doi">10.1038/s41591-021-01312-x</pub-id><pub-id pub-id-type="medline">33820998</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Han</surname><given-names>R</given-names> </name><name name-style="western"><surname>Acosta</surname><given-names>JN</given-names> </name><name name-style="western"><surname>Shakeri</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Ioannidis</surname><given-names>JPA</given-names> </name><name name-style="western"><surname>Topol</surname><given-names>EJ</given-names> </name><name name-style="western"><surname>Rajpurkar</surname><given-names>P</given-names> </name></person-group><article-title>Randomised controlled trials evaluating artificial intelligence in clinical practice: a scoping review</article-title><source>Lancet Digit Health</source><year>2024</year><month>05</month><volume>6</volume><issue>5</issue><fpage>e367</fpage><lpage>e373</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(24)00047-5</pub-id><pub-id pub-id-type="medline">38670745</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ng</surname><given-names>MY</given-names> </name><name name-style="western"><surname>Youssef</surname><given-names>A</given-names> </name><name name-style="western"><surname>Miner</surname><given-names>AS</given-names> </name><etal/></person-group><article-title>Perceptions of data set experts on important characteristics of health data sets ready for machine learning: a qualitative study</article-title><source>JAMA Netw Open</source><year>2023</year><month>12</month><day>1</day><volume>6</volume><issue>12</issue><fpage>e2345892</fpage><pub-id pub-id-type="doi">10.1001/jamanetworkopen.2023.45892</pub-id><pub-id pub-id-type="medline">38039004</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Checklist 1</label><p>DELPHISTAR reporting guidelines checklist.</p><media xlink:href="jmir_v28i1e91472_app1.docx" xlink:title="DOCX File, 229 KB"/></supplementary-material></app-group></back></article>