<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e97131</article-id><article-id pub-id-type="doi">10.2196/97131</article-id><article-categories><subj-group subj-group-type="heading"><subject>Viewpoint</subject></subj-group></article-categories><title-group><article-title>From Innovation to Impact: The CREATE Framework as a Blueprint for Large Language Model Adoption in Opioid Treatment Programs</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Markatou</surname><given-names>Marianthi</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Mukhopadhyay</surname><given-names>Raktim</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Good</surname><given-names>Jeff</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Tan</surname><given-names>Yihao</given-names></name><degrees>MA</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Dharia</surname><given-names>Arpan</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Brown</surname><given-names>Lawrence S</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Talal</surname><given-names>Andrew H</given-names></name><degrees>MD, MPH</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Biostatistics, SPHHP, University at Buffalo, State University of New York</institution><addr-line>Kimball Tower</addr-line><addr-line>Buffalo</addr-line><addr-line>NY</addr-line><country>United States</country></aff><aff id="aff2"><institution>Department of Linguistics, CAS, University at Buffalo, State University of New York</institution><addr-line>Buffalo</addr-line><addr-line>NY</addr-line><country>United States</country></aff><aff id="aff3"><institution>Department of AI and Society, CAS, University at Buffalo, State University of New York</institution><addr-line>Buffalo</addr-line><addr-line>NY</addr-line><country>United States</country></aff><aff id="aff4"><institution>Division of Gastroenterology, Hepatology and Nutrition, JSMBS, University at Buffalo, State University of New York</institution><addr-line>Buffalo</addr-line><addr-line>NY</addr-line><country>United States</country></aff><aff id="aff5"><institution>Department of Public Health, Weill Cornell Medicine</institution><addr-line>New York</addr-line><addr-line>NY</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Chatzimina</surname><given-names>Maria</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Singhal</surname><given-names>Mohit</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Marianthi Markatou, PhD, Department of Biostatistics, SPHHP, University at Buffalo, State University of New York, Kimball Tower, Buffalo, NY, 14214, United States, 1 7168292894; <email>markatou@buffalo.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>7</day><month>10</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e97131</elocation-id><history><date date-type="received"><day>03</day><month>04</month><year>2026</year></date><date date-type="rev-recd"><day>02</day><month>09</month><year>2026</year></date><date date-type="accepted"><day>02</day><month>09</month><year>2026</year></date></history><copyright-statement>&#x00A9; Marianthi Markatou, Raktim Mukhopadhyay, Jeff Good, Yihao Tan, Arpan Dharia, Lawrence S Brown, Andrew H Talal. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 7.10.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e97131"/><abstract><p>Large language model (LLM)&#x2013;based systems have tremendous potential to improve patient-centered health care, especially for medically underserved populations. However, realizing this potential requires careful consideration of the socio-technical contexts in which LLMs are used. The design of such systems should consider the vulnerabilities of any medically underserved group that the system intends to support, and provide trustworthy evidence for its use. This viewpoint reports the lessons learned from our experience and research with the CREATE (Culture, Respect, Education, Advancement, Trust, and Expertise) framework for engaging multiple stakeholders to guide the integration of LLM-based systems for improved health care delivery. We propose an extension of the traditional hierarchy of evidence model and identify key junction points in clinical workflows at which LLM-based systems can potentially be used to improve patient care. The key messages can be summarized as follows: (1) users of AI technologies, such as LLM-based tools, must be aware of the strengths, limitations, and impact of these technologies on health care delivery workflows and on the quality of generated evidence on which health care recommendations are based; (2) users must also be aware of the impact of data quality on AI outputs and on the evidence generated from the use of these systems; (3) the evidence pyramid provides a framework that facilitates the evaluation of the output generated by AI systems; (4) the CREATE framework facilitates the engagement of a variety of stakeholders. We propose concrete approaches for its validation and implementation; (5) any system designed to support people with opioid use disorder (OUD) needs to consider the overall lack of trust and stigmatizing experiences of this population with the health care system, and we discuss various aspects of the evaluation process necessary to build trust; (6) to discuss research directions that need to be addressed before LLM-based systems can be integrated usefully in patient health care.</p></abstract><kwd-group><kwd>Large language models</kwd><kwd>medically underserved populations</kwd><kwd>opioid use disorder</kwd><kwd>evidence generation</kwd><kwd>evaluation</kwd><kwd>risk</kwd><kwd>engagement</kwd><kwd>CREATE framework</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Human history has been marked by technological innovation. Generative AI (GenAI), including large language models (LLMs), has resulted in a technological revolution [<xref ref-type="bibr" rid="ref1">1</xref>]. While LLMs have tremendous technological potential, we, as a society, must be cognizant of the enormous potential social costs of LLMs without deliberate inclusiveness and safeguards. Failure to include the entire population, including medically underserved populations, in the AI technological revolution risks increased wage disparities, reduced productivity, and lower government revenue. Medically underserved populations include geriatric individuals, those with learning disabilities, low-income individuals, homeless individuals, and those with substance use disorder. The effects of the digital divide, conventionally referred to as persistent health care disparities in digital health care access [<xref ref-type="bibr" rid="ref2">2</xref>], are most acutely felt in these populations. The disparities can lead to exclusion from the use of new technologies, such as AI, especially among underserved populations.</p><p>For technological innovations to be acceptable to and accessible to the public, we must consider both social and technical aspects. Examples of sociotechnical innovations in health care include electronic health records (EHRs), communication applications, clinical decision support, and telehealth models. The social aspect of these health care tools involves user workflows and organizational policies surrounding their use, which are influenced by individuals&#x2019; level of digital literacy.</p><p>To address the social aspects of the deployment of digital technologies, we adapted the CREATE framework proposed by Talal et al [<xref ref-type="bibr" rid="ref3">3</xref>]. Our adaptation reflects CREATE as (C=culture, R=respect, E=education, A=advancement, T=trust, E=expertise). The CREATE framework outlines the processes to engage stakeholders in the acceptability, feasibility, and usability of digital technologies in various communities. Implementation of new technologies must consider how individuals will engage with the technology. An engagement framework, such as CREATE, is necessary when considering people with opioid use disorder (OUD), a patient population that is typically stigmatized, medically underserved, and difficult to reach [<xref ref-type="bibr" rid="ref4">4</xref>]. People with OUD enroll in opioid treatment programs (OTPs) for medical and behavioral treatment of addiction recovery. The recovery journey can be difficult, lengthy, and requires dedicated OTP staff involvement to comprehensively manage patients&#x2019; progress. New York State was a pioneer in the treatment of OUD [<xref ref-type="bibr" rid="ref5">5</xref>]. As a result, its procedures for treatment of OUD have been honed over decades. Although in this paper we refer to New York State&#x2013;based OUD treatment procedures as a use case, substance use disorder treatment requires the collection of extensive clinical information. The lessons learned are transferable to the treatment of substance use disorder and health care more broadly.</p><p>In the treatment of OUD, one of the fundamental instruments that inform the creation of treatment plans in OTPs is the psychosocial evaluation form completed within 30 days of admission. The psychosocial form is a comprehensive and detailed history of critical life areas and includes, among other aspects, past and present drug and alcohol use history, prior treatment history, family, legal, and trauma history, and physical and mental health information. The collection of these data and the construction of a patient-centered treatment plan are one of the most important and time-consuming processes of a counselor&#x2019;s responsibilities. Enhancing the value of the psychosocial instrument by implementing new technologies in the workflows of OTPs has the potential to streamline data collection and mitigate inefficiencies. It might ensure that patient information is collected in a systematic, reproducible, and standardized manner, resulting in high-quality data for use in the development of treatment plans.</p><p>The pursuit of high-quality, data-driven evidence enhanced by information generated from complementary sources, such as high-quality qualitative studies, allows individuals to make better choices about health and health care [<xref ref-type="bibr" rid="ref6">6</xref>]. In the OTP setting, evidence is translated into trust. Before LLM-based systems can be trusted with personal health information, an important objective is to ensure that all stakeholders agree that such systems are error-free, rigorous, and reproducibly generate data-driven evidence. These systems can also be of assistance when collecting data necessary to generate evidence that ultimately leads to improved health outcomes and/or better health care delivery.</p><p>The evaluation of technology incorporated into OTP workflows consists of assessing the safety and process compliance of the tools as well as evaluating their effectiveness. Therefore, before embedding LLM-based systems into OTP workflows, a safety evaluation is necessary to avoid and/or correct errors and ensure compliance with institutional metrics and standards (process compliance) [<xref ref-type="bibr" rid="ref7">7</xref>]. An additional level of evaluation that is required is an assessment of effectiveness. This means that unambiguous and reproducible evidence is obtained that demonstrates improved outcomes. The goal of the evaluation process is a clear demonstration that the embedded technology has a positive impact on the health of individuals and/or the health care delivery process.</p><p>A risk-benefit assessment is important to compare potential risks against beneficial outcomes. Identified risks associated with the use of LLM-based systems in OTP workflows include potential biases, hallucinations (ie<italic>,</italic> confabulation by the LLM-based system), and loss of privacy, security, and confidentiality. Benefits include the potential to streamline administrative tasks, resulting in considerable time savings, and improve patient assessment and education. Some of these tasks tolerate the incurrence of small errors that can be corrected by the human interfacing with these systems, while others, such as patient assessment, must be error-free. The development of quantitative risk-benefit measures ensures informed decision-making. Regulatory processes need to ensure the initial safety of the AI tools with continuous monitoring to guarantee quality and adaptability to current circumstances.</p><p>The paper explores important social considerations and evidence generation needs when attempting to integrate LLMs into clinical spaces for health care delivery. Recent investigation within the health care arena has focused on LLMs promoting increased patient-centered health care [<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref9">9</xref>], partially through the expression of empathy [<xref ref-type="bibr" rid="ref9">9</xref>]. Therefore, the primary aims of this viewpoint are: (1) to elucidate the strengths and limitations of using AI technologies, such as LLM-based tools, in health care delivery workflows (this aim is articulated in the remainder of the paper). To provide recommendations for potential adoption in clinical scenarios with an emphasis on underserved populations to facilitate patient-centeredness (see sections &#x201C;What are Recommendations to Improve Applicability and Trust of LLMs to Vulnerable Populations?&#x201D; and &#x201C;Considerations for Deployment of AI Systems for Underserved Populations<italic>&#x201D;</italic>). To discuss the adoption of these systems through the lens of Diffusion of Innovations (DOI) theory [<xref ref-type="bibr" rid="ref8">8</xref>] (see &#x201C;LLMs as Socio-Technical Systems: Diffusion of Innovation<italic>&#x201D;</italic>). (2) To discuss the evaluation of LLM-based systems from multiple aspects (see &#x201C;Evaluation of LLM-based Systems<italic>&#x201D;</italic>). (3) To discuss the impact of data quality on AI outputs and evidence generated from these systems; and to discuss the impact of using the generated evidence on health care decision-making (see sections &#x201C;Use of LLM-based Systems for High-Quality Data Acquisition and Improving Clinical Outcomes,&#x201D; and &#x201C;Evidence Generation and Risk Assessment in OTPs&#x201D;). (4) To present and discuss an extension of the traditional hierarchy of evidence model to include the risks associated with deploying AI-based systems at various stages of data collection, exemplified in the case of OTPs (&#x201C;Evidence Generation and Risk Assessment in OTPs&#x201D;). (5) To propose the CREATE framework as a facilitator of the participation of all relevant stakeholders in the development and deployment of AI systems in health care workflows and present several considerations to gain acceptance of LLM-based systems by necessary stakeholders to maximize the likelihood of deployment and use (&#x201C;Engagement Approaches: CREATE Framework&#x201D;).</p></sec><sec id="s2"><title>LLMs: Definition, Advantages, and Limitations</title><p>According to the US Congress [<xref ref-type="bibr" rid="ref9">9</xref>], &#x201C;Large language models (LLMs) are AI systems that aim to model language, sometimes using millions or billions of parameters.&#x201D; On a more technical level, LLMs are complex neural networks using the transformer architecture and the attention mechanism proposed in [<xref ref-type="bibr" rid="ref10">10</xref>]. The word &#x201C;large&#x201D; usually refers to the large number of parameters (often billions of parameters) used in training the neural network model to process the language. LLMs require considerable computational resources to be trained. LLMs form the core of LLM-based systems designed to provide various kinds of functionality such as machine translation, summarization, and conversational systems. <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> presents a brief history of natural language processing (NLP) along with a timeline of evolution of the field of NLP (Figure S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), and <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref> briefly discusses details associated with LLMs and applications in health care [<xref ref-type="bibr" rid="ref11">11</xref>-<xref ref-type="bibr" rid="ref14">14</xref>].</p><p>With newer applications of LLMs emerging every day, it is imperative to be informed of the limitations of these systems as well. Several technical and social challenges are associated with deploying any LLM-based system and their use context. The technical challenges of such systems are discussed in detail in this article from the perspective of accuracy of the generated output and usability of such systems, which are evaluated primarily on output quality. Unfortunately, their usability is often overlooked in outcome assessment. We emphasize that usability of such systems, including their safety and reliability, is an important consideration because in clinical settings such systems do not exist in isolation but rather as integrated socio-technical systems (STSs) to facilitate patient care (see section &#x201C;Evaluation of LLM-based Systems&#x201D;).</p><p>Deploying LLM-based tools in health care might also lead to potential social challenges. The authors in [<xref ref-type="bibr" rid="ref15">15</xref>] describe various risks associated with using LLM-based tools in public health. These risks are even more prominent when serving an underserved population. Given that OTPs function as destigmatizing safe spaces, the implications of introducing LLM-based tools in this environment must be carefully evaluated. The key risks include erecting additional barriers to help-seeking, degradation of patient-provider trust and support systems, missed opportunities to proactively introduce help, and dehumanization and impersonality in care, among others. We provide detailed descriptions of various activities in which LLM-based systems can be used in OTPs while keeping in mind these risks and ensuring that patient-centeredness is not lost.</p></sec><sec id="s3"><title>LLMs as STSs: DOI</title><p>STS is defined as &#x201C;a complex system that includes both technical and social elements, such as people, technologies, rules, and regulations, that work together to create work processes and products&#x201D; [<xref ref-type="bibr" rid="ref16">16</xref>]. The definition of STS immediately points to the fact that LLM-based tools in the context of health care are essentially STSs. The conceptual model for health information technology (HIT) introduced by Sittig and Singh [<xref ref-type="bibr" rid="ref17">17</xref>] is appropriate for this context. The 8 dimensions of this model are computing infrastructure, clinical content (clinical &#x201C;data-information-knowledge continuum&#x201D;), human-computer interface, people, workflow and communication, internal organizational policies, procedures and culture, and external rules and regulations. This conceptual model is appropriate for an LLM-based system as well. In the context of an LLM-based system, the technical component is represented by the computing infrastructure, the model architecture, datasets, and the associated infrastructure required to build such systems. The social component includes system developers, patients, the institution where such a system is deployed, the clinical staff, as well as ethics and any associated regulations. Failure to consider the interdependencies between the different components has the potential to result in the pitfalls described below [<xref ref-type="bibr" rid="ref18">18</xref>].</p><list list-type="order"><list-item><p>Framing trap: this trap occurs when a system is deployed without completely understanding the associated social setting and details of the problem the tool aims to solve. The problem formulation needs to incorporate the social context such as people, institutional environments, decision-making processes, and existing regulatory systems.</p></list-item><list-item><p>Portability trap: this arises when technologies designed for different social contexts and purposes are transferred to new settings without considering characteristics of the new social setting. This can often lead to harmful consequences.</p></list-item><list-item><p>Formalism trap: the evaluation of an STS often requires considering the context in which such systems are used, without which the evaluation of a system is merely mathematical in nature.</p></list-item><list-item><p>Ripple effect trap: ripple effects occur due to unintended consequences of introducing new technology to existing social systems (such as an OTP in this case). This can lead to unintended changes in behaviors and values of the existing system.</p></list-item><list-item><p>Solutionism trap: the failure to recognize that the solution to a problem might not involve introducing new technology.</p></list-item></list><p>The effects of getting stuck in any of these traps in a health care setting are consequential for treatment outcomes and patient health. The successful integration of LLM-based systems into existing settings and workflows depends on understanding the abilities and limitations of the components of the STS, the LLM-based system itself, as well as the social context in which it is deployed. The CREATE framework described later in the paper (Section &#x201C;Engagement Approaches: CREATE Framework&#x201D;) emphasizes the social considerations needed to increase the likelihood of full integration of LLM-based systems into OTPs.</p><p>To avoid the pitfalls discussed previously, understanding the technical aspects of LLM-based systems alone is insufficient; one must also consider the social system context. We offer insights on the adoption of LLM-based tools in OTPs through the lens of DOI theory. DOI theory has been widely applied to describe the important considerations when introducing any new technology. The adoption of new technology within the health care context is always a multistakeholder approach, which extends beyond the potential technical advantages, considering multiple stakeholders and defining their role in the adoption process. While DOI describes how an intervention spreads within a setting, CREATE serves as a framework for consideration of the social aspects of intervention adoption. Of particular interest in the OTP setting is its resource-scarce nature and the underserved population it serves, leading to additional trade-off considerations between purported advantages and potential loss of patient-centeredness and straining limited resources. We have described in detail the elements and characteristics of innovation (LLM-based tools) in the context of OTPs in <xref ref-type="table" rid="table1">Tables 1</xref> and <xref ref-type="table" rid="table2">2</xref>. The DOI theory highlights that the adoption of LLM-based tools is inherently an interplay between the social and the technical components. The aspects of trust and risk associated with such STS are discussed in a later section in this paper.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>The elements of innovation adapted to large language model (LLM)&#x2013;based tools in the context of opioid treatment programs.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Attribute</td><td align="left" valign="top">Entity</td><td align="left" valign="top">Relevance</td></tr></thead><tbody><tr><td align="left" valign="top">Innovation</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>LLM-based tools to improve and aid in patient outcomes.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Focus on patient safety in a medically underserved population.</p></list-item><list-item><p>Efficiency and decision-making may improve patient outcomes.</p></list-item></list></td></tr><tr><td align="left" valign="top">Adopters</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Providers</p></list-item><list-item><p>Government &#x0026; third-party payers</p></list-item><list-item><p>Patients and patient advocates.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Clinical staff are crucial for technological integration into clinical workflows. Regulatory authorities can standardize adoption, maintain care quality, and ensure appropriate system use.</p></list-item><list-item><p>Patients and stakeholders provide feedback and ensure effective and ethical use.</p></list-item></list></td></tr><tr><td align="left" valign="top">Time</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>AI is in the early stages of adoption in the OTP<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> setting.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Patient safety is crucial for medically underserved populations.</p></list-item><list-item><p>Patients face stigma and other barriers.</p></list-item><list-item><p>Ensure patient safety through maintaining privacy &#x0026; data confidentiality, without harm.</p></list-item></list></td></tr><tr><td align="left" valign="top">Communication</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>System Development</p></list-item><list-item><p>Financial Engagement</p></list-item><list-item><p>Care Delivery</p></list-item><list-item><p>Shaping Policy</p></list-item></list></td><td align="left" valign="top">AI system developers with input from regulatory authorities &#x0026; clinical staff:<list list-type="bullet"><list-item><p>Effective communication is important during product development.</p></list-item><list-item><p>Regulatory and health care providers provide necessary inputs to integrate AI systems into existing workflows.</p></list-item></list><break/><underline>S</underline>taff&#x21D4;staff, staff&#x21D4;patients, patient&#x21D4;patient:<list list-type="bullet"><list-item><p>Communication between staff and patients is essential to build trust in the system.</p></list-item><list-item><p>Effectiveness of the systems depends on integration into existing workflows.</p></list-item></list><break/>Payers&#x21D4;treatment providers:<underline/><list list-type="bullet"><list-item><p>Reimbursement policies and support are needed for care provided using LLM-based tools.</p></list-item></list></td></tr><tr><td align="left" valign="top">Social System</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Government and Regulatory Authorities</p></list-item><list-item><p>Financial Stakeholders</p></list-item><list-item><p>Care Delivery Stakeholders</p></list-item><list-item><p>Professional Organizations and Advocacy Groups</p></list-item><list-item><p>Researchers</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Successful development, deployment, and use of socio-technical systems in real-world settings depend on the social system.</p></list-item><list-item><p>Engagement with the ecosystem facilitates continuous refinement through feedback.</p></list-item><list-item><p>The CREATE<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> framework (<xref ref-type="fig" rid="figure1">Figure 1</xref>) addresses the social aspects when integrating new technology into substance-use treatment programs.</p></list-item></list></td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>OTP: Opioid Treatment Program.</p></fn><fn id="table1fn2"><p><sup>b</sup>CREATE: Culture, Respect, Education, Advancement, Trust, Expertise.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Characteristics of innovation for large language model (LLM)&#x2013;based tools in opioid treatment programs. Adapted from [<xref ref-type="bibr" rid="ref8">8</xref>].</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Attributes</td><td align="left" valign="bottom">Relevance</td></tr></thead><tbody><tr><td align="left" valign="top">Relative advantage</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Improved efficiency: LLM-based tools reduce repetitive administrative tasks, freeing providers to focus on clinical tasks.</p></list-item><list-item><p>Enhanced decision-making: LLMs can process vast medical texts to aid diagnostic and clinical decision-making.</p><list list-type="bullet"><list-item><p>LLM-based systems assisting health care providers can potentially improve care quality and reduce errors.</p></list-item></list></list-item><list-item><p>An important consideration is whether these LLM-based systems should be patient-facing or used solely for internal workflows.</p><list list-type="bullet"><list-item><p>Potential applications: aid in filling out intake forms.</p></list-item><list-item><p>Risks include digital literacy in the OUD<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> population.</p></list-item><list-item><p>Furthermore, a benefit is the destigmatization that can occur. Patients may share sensitive information more comfortably without discrimination and shame occurring during human interactions.</p></list-item></list></list-item></list></td></tr><tr><td align="left" valign="top">Compatibility</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>New systems must be compatible with existing workflows.</p></list-item><list-item><p>Emphasis on usability. Usability focuses on learnability, efficiency, memorability, error prevention &#x0026; recovery, and user satisfaction.</p></list-item><list-item><p>Companies are integrating LLM-based solutions into EHR<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup> systems for various uses [<xref ref-type="bibr" rid="ref19">19</xref>].</p></list-item></list></td></tr><tr><td align="left" valign="top">Complexity</td><td align="left" valign="top">Applications can be grouped into one of the three groups:<break/>Low complexity<list list-type="bullet"><list-item><p>Simple tasks for patient communication are low complexity and have minimal associated risk.</p></list-item></list><break/>Medium Complexity<list list-type="bullet"><list-item><p>LLM-based systems used for patient education, such as answering queries, providing instructions, and offering health education.</p></list-item><list-item><p>Associated risk can be mitigated potentially through grounding the responses on a knowledge base.</p></list-item><list-item><p>The system should not be detrimental to human health.</p></list-item></list><break/>High Complexity<list list-type="bullet"><list-item><p>Examples include diagnostic and decision-making support.</p></list-item></list><break/>LLM-based systems must ensure safety, build trust, ensure usability, risk education, and effective training for staff and patients for seamless integration.</td></tr><tr><td align="left" valign="top">Trialability</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Most LLM-based applications are not presently implemented in OTPs<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup>, indicating an opportunity for adoption in this domain.</p></list-item><list-item><p>Predeployment trials are essential before deploying any AI system, especially in health care, where errors are unacceptable.</p><list list-type="bullet"><list-item><p>Community partnerships (eg, between OTPs and research universities) can facilitate these trials.</p></list-item><list-item><p>Active stakeholder engagement (eg, site visits and stakeholder engagement) is needed for extending, addressing, and facilitating large-scale implementation.</p></list-item></list></list-item></list></td></tr><tr><td align="left" valign="top">Observability</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Current implementations of LLM-based systems directed toward the OUD population or OTPs do not exist.</p></list-item><list-item><p>Evidence from other areas in medicine highlights the potential of LLM-based tools in improved patient-doctor communication [<xref ref-type="bibr" rid="ref20">20</xref>], generating patient-friendly and accurate notes [<xref ref-type="bibr" rid="ref20">20</xref>], enhanced diagnostic accuracy in pulmonology and endocrinology [<xref ref-type="bibr" rid="ref21">21</xref>], pathology report explanation [<xref ref-type="bibr" rid="ref22">22</xref>], medical exam recommendations and diagnosis [<xref ref-type="bibr" rid="ref23">23</xref>], clinical decision support systems [<xref ref-type="bibr" rid="ref24">24</xref>].</p></list-item></list></td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>OUD: opioid use disorder.</p></fn><fn id="table2fn2"><p><sup>b</sup>EHR, electronic health record.</p></fn><fn id="table2fn3"><p><sup>c</sup>OTP: Opioid Treatment Program.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>CREATE (C=culture, <italic>R</italic>=respect, E=education, A=advancement, T=trust, E=expertise) framework for engaging stakeholders in the integration of AI within health care.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e97131_fig01.png"/></fig></sec><sec id="s4"><title>Engagement Approaches: CREATE Framework</title><p>To guide engagement of stakeholders with LLM-based systems, we extended the CREATE framework developed in [<xref ref-type="bibr" rid="ref3">3</xref>] (<xref ref-type="fig" rid="figure1">Figure 1</xref>). The CREATE framework outlines the processes to address the social aspects of technological integration into a community. Elucidating clinical workflows for patient care facilitates understanding of the OTP culture. Respect for OTP staff and patients is manifested by incentives, avoiding stigmatizing language, and through co-leadership decisions. Education about technology, addressing anxiety through knowledge, and co-learning between patients and staff promote a respectful attitude toward technology. Education also promotes technological advancement through invention and the application of new tools. Promoting trust among patients and staff in LLM-based tools is required for deployment. LLM use is enhanced by ensuring privacy, confidentiality, and security of the technology. Successful LLM deployment and use require expertise in multiple disciplines, and the CREATE framework can address the social aspects of technological deployment in underserved communities. Considerations for evaluation matrices of each of the 6 CREATE framework domains are illustrated in Table S1 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>. Consideration of the CREATE framework domains, as judged against conventional implementation frameworks, is listed in Table S2 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p><p>As a major funder of comparative effectiveness research, the Patient-Centered Outcomes Research Institute (PCORI) has developed 6 foundational aspects of research partnerships. <xref ref-type="table" rid="table3">Table 3</xref> illustrates how the CREATE framework aligns with PCORI&#x2019;s 6 foundational expectations for partnerships in research [<xref ref-type="bibr" rid="ref25">25</xref>].</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Foundational aspects of research partnership.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Category</td><td align="left" valign="top">Definition</td><td align="left" valign="top">Relevance to CREATE<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup> framework</td></tr></thead><tbody><tr><td align="left" valign="top">Representative involvement</td><td align="left" valign="top">Inclusion of partners and organizations that reflect a range of affected patients and communities.</td><td align="left" valign="top">Multidomain expertise is required for engagement success and promoting advancement.</td></tr><tr><td align="left" valign="top">Build capacity to work as a team</td><td align="left" valign="top">Identify strengths and obstacles to engagement; provide education to address obstacles.</td><td align="left" valign="top">Understanding culture and education can identify engagement opportunities; obstacles are overcome with education.</td></tr><tr><td align="left" valign="top">Early and ongoing engagement</td><td align="left" valign="top">Responsible party engaged throughout the technology lifecycle</td><td align="left" valign="top">Understanding culture illustrates how and where to engage.</td></tr><tr><td align="left" valign="top">Ongoing review and assessment of engagement</td><td align="left" valign="top">Perform continuous assessment and evaluation of successes and failures.</td><td align="left" valign="top">CREATE provides an overview of how engagement might be bolstered or modified</td></tr><tr><td align="left" valign="top">Meaningful inclusion of partners in decision-making</td><td align="left" valign="top">Use approaches to include multiple partners in decision-making</td><td align="left" valign="top">Principle promotes co-learning and multidomain expertise.</td></tr><tr><td align="left" valign="top">Dedicated funds for engagement and partner compensation</td><td align="left" valign="top">Resources to compensate for time and effort.</td><td align="left" valign="top">Funds can promote respect and trust as CREATE components.</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>CREATE: C=culture, R=respect, E=education, A=advancement, T=trust, E=expertise, explains social aspects of technology deployment in a community. </p></fn></table-wrap-foot></table-wrap></sec><sec id="s5"><title>What Are Some of the Functions of LLM-Based Systems in OTPs?</title><p>LLM-based systems have shown good performance in several use cases. Some examples include responding to prevention questions from patients in cardiology [<xref ref-type="bibr" rid="ref26">26</xref>], hip replacement [<xref ref-type="bibr" rid="ref27">27</xref>], and radiology report findings [<xref ref-type="bibr" rid="ref28">28</xref>]. Novel experiments integrating LLMs as clinical decision support tools are needed to evaluate their effect on outcomes, productivity, and patient satisfaction [<xref ref-type="bibr" rid="ref29">29</xref>]. Existing literature has evaluated LLMs&#x2019; abilities to respond to portal patient messages [<xref ref-type="bibr" rid="ref30">30</xref>], generate discharge summaries [<xref ref-type="bibr" rid="ref31">31</xref>], generate structured templates for radiology [<xref ref-type="bibr" rid="ref28">28</xref>], and support the management of breast cancer tumor boards [<xref ref-type="bibr" rid="ref32">32</xref>]. Furthermore, LLM-based systems have been successfully implemented in diagnostic image processing and clinical decision support [<xref ref-type="bibr" rid="ref33">33</xref>]. They play a role in social media monitoring and track trends in OUD that correlate with real-time morbidity and mortality reporting [<xref ref-type="bibr" rid="ref34">34</xref>]. AI can also be useful in the development of medical guidelines [<xref ref-type="bibr" rid="ref35">35</xref>] and in the construction of systematic reviews [<xref ref-type="bibr" rid="ref36">36</xref>].</p><p>Part of the challenge of integrating LLM-based systems into clinical workflows is that minor changes in the inputs produce unpredictable outputs [<xref ref-type="bibr" rid="ref29">29</xref>]. Furthermore, the technology is evolving rapidly, and as it becomes more widespread, a potential issue is overreliance on its output while negating its limitations and biases. As an alternative to AI in the application to health care, the American College of Physicians recommends the term &#x201C;augmented intelligence&#x201D; when referring to the role of AI in clinical decision-making. The objective is to promote the concept that human intelligence continues to be central to the clinical decision-making process even when integrating AI, and that AI is a tool to assist clinicians [<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref38">38</xref>].</p><p>LLM-based systems can be used to address the digital divide. We can use knowledge generation and educational functionalities to enhance digital literacy among patients. LLM-based systems can also be used to combine data on a particular patient within an electronic medical record and to support prior authorizations for medications. In these situations, it would be helpful to have the human in the loop [<xref ref-type="bibr" rid="ref39">39</xref>] to be able to verify information. Human-in-the-loop (HITL) refers to the integration of an expert into a machine learning or AI workflow to improve the quality of the system&#x2019;s intended outcomes. In the health care context, expert knowledge and experience are essential, requiring the augmentation of skilled experts, rather than subordinating their skills. Similar to HITL, the terminology &#x201C;Doctor-in-the-loop&#x201D; has also been coined [<xref ref-type="bibr" rid="ref40">40</xref>]. Several works [<xref ref-type="bibr" rid="ref41">41</xref>] have explored the incorporation of expert knowledge to improve the trustworthiness of AI system&#x2013;based workflows. In health care, transparency is one of the key requirements for LLM-based systems, which can then support the HITL framework to increase trust in the system [<xref ref-type="bibr" rid="ref42">42</xref>].</p><p>Appropriate LLM-based systems might promote patient-centeredness. LLMs can de-stigmatize language and can suggest alternative language choices. The goal is to promote more empathetic language among health care providers and in public health messaging [<xref ref-type="bibr" rid="ref43">43</xref>]. LLMs can potentially play an important role in patient and staff education. They can develop culturally and linguistically relevant educational materials, potentially improving patient understanding, medication adherence, and trust in providers [<xref ref-type="bibr" rid="ref44">44</xref>]. LLMs, in combination with structured EHR data, can develop models that predict treatment attrition, enabling targeted interventions [<xref ref-type="bibr" rid="ref45">45</xref>]. They may have a potential role in OTP staff training and simulation. LLM-driven virtual patient agents can simulate realistic patient-provider interactions, offering a controlled and ethical environment for training OTP staff in therapeutic dialogue strategies for addiction recovery [<xref ref-type="bibr" rid="ref46">46</xref>]. The goal is to promote LLM-directed patient-centered messaging and education.</p></sec><sec id="s6"><title>Potential Use of LLM-Based System in Treatment Plan Development and Implementation</title><p>As presently performed, the generation of a client&#x2019;s treatment plan is a relatively lengthy and laborious process as described in <xref ref-type="fig" rid="figure2">Figure 2.1</xref>. When a patient is admitted to an OTP in New York State, the first point of contact is an initial phone screen that gathers basic demographic information and assesses patient eligibility for treatment of OUD. The next step is an admission eligibility assessment by a physician or advanced practice provider. Once the patient is admitted to the OTP, the clinician completes an admission form that includes more extensive information on drug use, residential situation, family history, and information on employment and education. The patient is then assigned to a primary counselor within 24 hours who schedules an initial session to obtain the psychosocial form, which must be completed within 30 days of admission (email, personal communication, Office of Addiction Services and Supports [OASAS], New York, March 31, 2026).</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>(1) Inputs to and Development of Treatment Plan: The steps for the development of the treatment plan are outlined in the text. Yellow highlight indicates opportunities for task coordination with large language models. (2) Extension of <xref ref-type="fig" rid="figure2">Figure 2.1</xref>, which highlights the key junctures where AI-based systems can be leveraged in the day-to-day Opioid Treatment Program workflow. Additionally, we have emphasized certain aspects of AI-based systems that need to be considered before such systems are implemented in real-world settings. For each distinct phase, we illustrate the role of the AI-based system in the blue box. The interaction of various components of AI-enhanced Opioid Treatment Programs is depicted using arrows. Each stage of the workflow can have different extents of patient centeredness embedded within it depending on the patient needs. LLM: large language model; OASAS: Office of Addiction Services and Supports.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e97131_fig02.png"/></fig><p>Once the psychosocial form has been completed, data are aggregated from the 4 sources illustrated in <xref ref-type="fig" rid="figure2">Figure 2.1</xref> to develop an assessment summary. Upon completion of the assessment summary, the multidisciplinary team meets to develop a treatment plan. The next step is to present and discuss the treatment plan with the patient. Following the discussion with the patient, an initial 90-day treatment course is implemented.</p><p>There are several steps at which LLM-based systems can be involved during the creation and implementation of the treatment plan, as illustrated in <xref ref-type="fig" rid="figure2">Figure 2.2</xref>. LLM-based systems can be integrated into the data collection stage to facilitate the collection of higher-quality data to inform patient care. Additionally, LLM-based systems can be used to aggregate the collected data from different sources with varied structures to develop treatment plans that consider the unique circumstances of a patient, prioritizing specific aspects of the treatment and supporting ongoing addiction management. Finally, such systems can be used to aid treatment adherence, which is known to be a difficult challenge in the case of addiction recovery. Important considerations for integration of LLM-based systems in different areas of the workflow are also presented.</p><p>Furthermore, an empathetic LLM-based system could potentially administer the psychosocial assessment and develop its clinical summary. LLMs could also develop the assessment summary through the aggregation of data from multiple sources. In terms of developing a treatment plan, LLMs could prioritize the most important aspects and assess their severity. The multidisciplinary team could ultimately assess and verify the output of the LLM-based system, similar to the HITL approach described previously. Finally, the LLM-based system can support patient adherence to the 90-day treatment plan.</p></sec><sec id="s7"><title>Use of LLM-Based Systems for High-Quality Data Acquisition and Improving Clinical Outcomes</title><p>In the previous section, we have illustrated how LLM-based systems can be used for developing individualized treatment plans for patients with OUD. One of the major components of developing these treatment plans is collecting data using high-quality instruments. The goal of using LLM-based systems is to generate individualized treatment plans dedicated to specific patient types and reduce associated errors.</p><p>From a higher level of abstraction, an important research interest is how LLM-based systems or LLMs can be used for processing vast amounts of data to inform better patient care. OTPs collect massive amounts of textual data consisting of clinical narratives, patient-reported information, and other records. LLMs can be used for converting free text into structured datasets, performing data extraction, and other data processing tasks. These procedures broadly aid in dataset generation that informs care about specific populations.</p><p><xref ref-type="table" rid="table4">Table 4</xref> lists the various proposed applications of LLM-based systems along with associated evidence of their implementation in health care situations and briefly indicates whether HITL might be required for oversight.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Various proposed applications of large language model (LLM)&#x2013;based systems in health care along with associated evidence and need for human-in-the-loop oversight.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Proposed application using LLM-based system</td><td align="left" valign="bottom">Evidence</td><td align="left" valign="bottom">HITL<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup> required (Yes or no)</td></tr></thead><tbody><tr><td align="left" valign="top">Patient portal messaging</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref30">30</xref>]</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Scheduling does not require HITL.</p></list-item><list-item><p>Health care inquiries need HITL validation.</p></list-item></list></td></tr><tr><td align="left" valign="top">Clinical decision support systems</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref48">48</xref>]</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Yes</p></list-item></list></td></tr><tr><td align="left" valign="top">Destigmatizing language and suggesting empathetic alternatives</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref49">49</xref>]</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Yes</p></list-item></list></td></tr><tr><td align="left" valign="top">Patient and staff education using relevant materials</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref11">11</xref>]</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Partial. HITL is needed for verification of information accuracy and updating of knowledge base.</p></list-item></list></td></tr><tr><td align="left" valign="top">Predicting treatment, attrition &#x0026; targeted interventions</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref45">45</xref>]</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Yes</p></list-item></list></td></tr><tr><td align="left" valign="top">Patient simulation</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref46">46</xref>]</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Partial. HITL is needed for updating and verification.</p></list-item></list></td></tr><tr><td align="left" valign="top">Administering intake questionnaires</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref50">50</xref>]</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Yes</p></list-item></list></td></tr><tr><td align="left" valign="top">Generating clinical assessment summaries</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref51">51</xref>]</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Yes</p></list-item></list></td></tr><tr><td align="left" valign="top">Free text to structured report generation</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref52">52</xref>]</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Yes</p></list-item></list></td></tr><tr><td align="left" valign="top">Developing treatment plans</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref54">54</xref>]</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Yes</p></list-item></list></td></tr><tr><td align="left" valign="top">Supporting patient adherence to treatment plans</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref55">55</xref>]</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Dependent on the magnitude of the task. For simple and routine verification of adherence, HITL is not required.</p></list-item></list></td></tr><tr><td align="left" valign="top">Enhancing digital literacy</td><td align="left" valign="top">Proposed</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Partial</p></list-item></list></td></tr><tr><td align="left" valign="top">Data aggregation from various sources to produce a new dataset</td><td align="left" valign="top">Proposed</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Yes</p></list-item></list></td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>HITL: human-in-the-loop.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s8"><title>Evaluation of LLM-Based Systems</title><p>Prior to the integration of any LLM-based systems, or rather any system in the health care context, understanding associated challenges and evaluating risks vs benefits is of utmost importance. These challenges impact the reliability of generated output, and consequently the quality of evidence and hence need to be addressed to ensure safety and trust. A summary of challenges in the health care context is given in <xref ref-type="table" rid="table5">Table 5</xref>.</p><p>The challenges discussed in this section necessitate a robust evaluation strategy considering multiple aspects described in the next section.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Key challenges, descriptions, and potential mitigation strategies for use of large language model (LLM)&#x2013;based systems in health care.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Challenges</td><td align="left" valign="bottom">Description</td><td align="left" valign="bottom">Mitigation and solution</td></tr></thead><tbody><tr><td align="left" valign="top">Hallucinations</td><td align="left" valign="top">Producing outputs that are factually incorrect and misleading [<xref ref-type="bibr" rid="ref56">56</xref>]. Divided into factual (discrepancy between generated output and real-world information) or faithfulness (discrepancy between generated output and provided instructions). These issues arise due to training data with outdated or false information [<xref ref-type="bibr" rid="ref57">57</xref>]. Validation fatigue and optimization for answers that look correct make detecting hallucinations difficult [<xref ref-type="bibr" rid="ref58">58</xref>].</td><td align="left" valign="top">Retrieval augmented generation (RAG), where responses are based on external knowledge bases [<xref ref-type="bibr" rid="ref59">59</xref>]; several techniques described in [<xref ref-type="bibr" rid="ref56">56</xref>].</td></tr><tr><td align="left" valign="top">Sycophancy</td><td align="left" valign="top">The tendency of LLMs to excessively agree with users [<xref ref-type="bibr" rid="ref60">60</xref>].</td><td align="left" valign="top">Implementing guardrails, ie, controlling the output of an LLM to respect some human-imposed constraints [<xref ref-type="bibr" rid="ref61">61</xref>].</td></tr><tr><td align="left" valign="top">Black-box nature</td><td align="left" valign="top">Poses a significant challenge to explainability of the model outputs.</td><td align="left" valign="top">Potential techniques for improving explainability for LLMs can be found in [<xref ref-type="bibr" rid="ref62">62</xref>].</td></tr><tr><td align="left" valign="top">Data protection</td><td align="left" valign="top">Systems must strictly adhere to global data protection laws to ensure patient privacy and confidentiality are not compromised.</td><td align="left" valign="top">Strict adherence to laws like Health Insurance Portability and Accountability Act (HIPAA) and General Data Protection Regulation (GDPR).</td></tr><tr><td align="left" valign="top">Nondeterminism</td><td align="left" valign="top">Poses a serious threat to reproducibility. Arises primarily due to a combination of floating-point non-associativity and parallel execution [<xref ref-type="bibr" rid="ref63">63</xref>].</td><td align="left" valign="top">Quantifying the uncertainty associated with LLM-based tools is essential for building trust and consequently their large-scale adoption and use in health care. Potential research in these directions is discussed in [<xref ref-type="bibr" rid="ref64">64</xref>-<xref ref-type="bibr" rid="ref66">66</xref>].</td></tr></tbody></table></table-wrap></sec><sec id="s9"><title>Multidimensional Evaluation of LLM-Based Systems</title><sec id="s9-1"><title>Overview</title><p>The American Medical Informatics Association (AMIA) recommends evaluation of AI systems in health care through multiple lenses, namely technical performance, health impact (effectiveness), and usability and workflows (operational efficiency) [<xref ref-type="bibr" rid="ref67">67</xref>]. This emphasizes that evaluation of LLM-based systems must not be based on technical aspects alone; rather, it must account for the STS in which it operates.</p><p>We describe these evaluation aspects below for clarity. We would like to note that these aspects are not isolated; rather, they are interdependent. For the purposes of this article, we consider usability under technical evaluation, but it is an overlapping concept that links the technical implementation of the system with its adoption and hence effectiveness.</p></sec><sec id="s9-2"><title>Technical Evaluation</title><p>The technical evaluation should consider the following factors affecting LLM outputs:</p><list list-type="order"><list-item><p>Parameters: LLM outputs depend on several user-defined parameters such as temperature, top-p, frequency penalty, presence penalty, and thinking or reasoning levels, which in turn affect the output. For instance, how does a set of parameters, for example, &#x03B8;=[temperature, thinking level, frequency penalty], impact the output O? Understanding the impact of these parameters is an important consideration for our work. Prompting also plays an important role in the output of LLM-based systems. Various prompting techniques, such as described in [<xref ref-type="bibr" rid="ref68">68</xref>], need to be investigated depending on the task.</p></list-item><list-item><p>Repeatability and faithfulness: few works [<xref ref-type="bibr" rid="ref69">69</xref>] have explored concepts such as:</p><list list-type="alpha-lower"><list-item><p>Semantic repeatability: the consistency of LLM responses across repeated runs under identical conditions.</p></list-item><list-item><p>Semantic reproducibility: the consistency across runs under different conditions.</p></list-item></list></list-item><list-item><p>HITL validation approaches are needed to ensure that evidence generated by LLM-based systems is accurate.</p></list-item><list-item><p>Usability evaluation: the definition of usability as given in ISO 9241&#x2010;11 is &#x201C;The extent to which a product can be used by specified users to achieve specified goals with effectiveness, efficiency, and satisfaction in a specified context of use&#x201D; [<xref ref-type="bibr" rid="ref70">70</xref>]. Usability evaluation of LLM-based systems is needed to understand whether the tool is practical, intuitive, and readily adoptable by clinical staff without introducing significant workflow hurdles or cognitive burdens. Several usability assessment techniques are available such as the Nielsen usability checklist, questionnaires, interviews, and observational studies, the System Usability Scale, usability testing, cognitive walkthrough, etc. Dehghani et al [<xref ref-type="bibr" rid="ref71">71</xref>] present a review of methods used in usability evaluation of hospital information systems.</p></list-item><list-item><p>Computational reproducibility for provenance: computational reproducibility is defined [<xref ref-type="bibr" rid="ref72">72</xref>] as ensuring that the same results using the exact same raw materials, computational steps, codes, and conditions as the original analysis can be obtained, and is the foundation for system provenance. In the context of LLM-based health care systems, provenance requires recording and tracking the entire lifecycle of an input (eg, patient data) and its corresponding output (eg, a treatment recommendation). Provenance in case of health care systems is extremely important for legal and regulatory accountability as well as for ensuring that the system behaves as intended. LLM-based systems should be able to trace back how a certain output was generated. LLM-based systems should be developed while keeping this in mind, and in turn, must be evaluated based on it as well.</p></list-item><list-item><p>Postdeployment monitoring: an analogue of postdeployment monitoring is postmarketing surveillance of medical products, which emphasizes the need to monitor medical products once they have been placed on the market following rigorous clinical trials. Similarly, to understand the performance of an LLM-based system and address potential shortcomings in real-world settings, postdeployment monitoring is of crucial importance to ensure that the system supports its intended users for its intended purpose. When LLM-based systems are deployed as a component of clinical workflows, several additional mechanisms are required simultaneously to ensure their holistic integration. Several frameworks have been proposed emphasizing various aspects of AI governance [<xref ref-type="bibr" rid="ref73">73</xref>-<xref ref-type="bibr" rid="ref75">75</xref>]. Regarding postdeployment monitoring, Keyes et al [<xref ref-type="bibr" rid="ref73">73</xref>] propose 3 complementary principles: system integrity, performance, and impact. System integrity addresses information technology components, such as maximizing system uptime, detecting runtime errors, and mitigating unintended consequences. Activities essential for performance monitoring include logging, auditing, evaluating bias, accuracy, predictability, transparency, and version control. Impact monitoring focuses on evaluating the value of the system and may lead to recalibration, rollback, suspension, or retirement when deployed systems no longer perform adequately. Additional administrative and institutional policies are needed for incident management and response. El Arab et al [<xref ref-type="bibr" rid="ref75">75</xref>] found that trustworthy AI implementation in health care settings is increasingly associated with continuous &#x201C;lifecycle governance&#x201D; as compared to predeployment validation only. Lifecycle governance is also a cornerstone of continuous quality improvement, which represents iterative improvement of &#x201C;processes, safety, and patient care&#x201D; [<xref ref-type="bibr" rid="ref76">76</xref>]. However, there is a lack of formal consensus, and practical challenges remain regarding metric selection, thresholds for action, review frequency, corrective response, and institutional accountability. These activities align with guidance from World Health Organization (WHO), which recommends that governments should introduce mandatory postdeployment auditing and impact assessments [<xref ref-type="bibr" rid="ref77">77</xref>].</p></list-item></list><p>Additional aspects of technical evaluation should also be considered, keeping in mind the context in which the system is deployed. For example, whether the output adheres to specific guardrails enforced in the system.</p></sec><sec id="s9-3"><title>Operational Efficiency (Workflows)</title><p>The objective for implementing any LLM-based system in a health care setting is to improve operational efficiency without compromising patient care. Given the administrative workload in OTPs, the system must not introduce additional cognitive overload as discussed in the usability evaluation. The LLM-based system should be evaluated on its ability to streamline workflows. Efficiency in this context is measured by how well the tool reduces the time and cognitive resources of clinical staff in accomplishing a task as compared to currently used solutions (baseline).</p></sec><sec id="s9-4"><title>Effectiveness</title><p>The implementation of any new system must prioritize and enhance patient care and/or health care delivery. In an OTP setting where LLM-based systems could be used to assist or potentially develop treatment plans, an error-free, highly accurate, and efficient performance is a requirement.</p><p>Evaluating health care delivery workflows involves analyzing clinical and administrative processes with the goal of identifying bottlenecks, reducing errors, and ultimately enhancing clinical care. The goal is to evaluate the impact of LLM-based workflows on health care outcomes. Several methods that evaluate technology-enhanced workflows, for example, the use of EHRs, exist in the literature. In the OTP context, if we desire to compare the effectiveness of LLM-based versus standard workflows, we could propose a noninferiority clinical trial using a biological outcome, for example, quantity of opioids measured in urine (ie, toxicology results). In designing trials of clinical effectiveness, one needs to understand the clinical workflows where LLM-based systems will be implemented and the factors beyond the technical capabilities of LLMs to bridge the gaps between implementation and adoption by OTP staff.</p></sec></sec><sec id="s10"><title>Evidence Generation and Risk Assessment in OTPs</title><p>Given the potential benefits and risks of implementing LLM-based systems in health care programs, decisions about their adoption must be guided by a balanced assessment of both dimensions. Beyond the general concerns discussed in prior sections, levels of risk of LLM integration in a health care program should be assessed critically based on the function of the model and the type of input data. For a better visualization of the benefits and risks aspects of LLM-based systems, the pyramid model presented in <xref ref-type="fig" rid="figure3">Figure 3</xref> illustrates the central rationale for LLM integration in OTPs. The deeper the integration of an LLM model with more clinically dedicated tasks and the richer the data sources, the higher the level of evidence obtained, and consequently, the higher the risk associated with LLM-based systems (<xref ref-type="fig" rid="figure3">Figure 3</xref>).</p><p>Telephone screening is the foundation of the pyramid, which typically occurs before formal OTP admission. This interaction involves acquiring basic demographic information and assessing patient eligibility for OUD treatment. The LLM-based system functions (eg, summarizing call transcripts, extracting key variables, and generating reminders for follow-up) work with the lowest level of input. As a result, errors from LLM-based systems at this stage rarely affect immediate clinical decisions, so the risk is minimal. Thus, we obtain the lowest level of evidence generation (&#x2013;2) in the pyramid model. During OTP admission, programs collect patient information on psychosocial forms. The pyramid illustrates that, at each subsequent level of risk, the complexity of clinical information increases. Simultaneously, the risks and consequences such as a breach become increasingly costly.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>A multifaceted pyramid of data, evidence generation, and risk associated with AI systems in an Opioid Treatment Program setting. The front face, representing data, highlights the hierarchy of data available at Opioid Treatment Programs for opioid use disorder treatment. When an opioid use disorder patient is first admitted, a screening phone call is conducted, acquiring preliminary data (lowest quality). Gradually, as the patient is incorporated into the Opioid Treatment Program, collection and aggregation of additional sources of data take place, which leads to a higher quality of data. Corresponding to each level of data quality, there is an associated level of evidence generation (left face). The right face of the pyramid represents the risk associated with deploying AI-based systems in each stage of data collection or aggregation, depicted in the front face. HCP: health care provider.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e97131_fig03.png"/></fig></sec><sec id="s11"><title>Regulatory Aspects</title><p>Policy frameworks for AI are currently in a state of rapid and continuous change. Regulations and laws written to govern the use of AI neither address the challenges nor the opportunities associated with such systems [<xref ref-type="bibr" rid="ref77">77</xref>]. In July 2024, the European Union adopted Regulation (EU) 2024/1689, commonly known as the EU Artificial Intelligence Act, thereby establishing the first AI regulation as a comprehensive framework to regulate AI systems [<xref ref-type="bibr" rid="ref78">78</xref>]. The act includes a risk-based regulatory framework that differentiates obligations for AI developers and deployers according to the level of risk posed by specific systems [<xref ref-type="bibr" rid="ref78">78</xref>-<xref ref-type="bibr" rid="ref80">80</xref>], categorizing the risk into unacceptable risk, high risk, transparency risk, and minimal risk. For further information see [<xref ref-type="bibr" rid="ref78">78</xref>].</p><p>In the United States, there are no federal laws dedicated solely to AI. The existing framework relies on a patchwork of device, data privacy (such as the Health Insurance Portability and Accountability Act [HIPAA]), and liability laws. However, the regulatory landscape for AI in the United States is currently undergoing a period of rapid and significant transition [<xref ref-type="bibr" rid="ref81">81</xref>, <xref ref-type="bibr" rid="ref82">82</xref>]. In December 2025, an executive order was issued to foster a more unified national environment for AI by aligning federal and state policies to ensure a supportive regulatory framework for AI innovation [<xref ref-type="bibr" rid="ref83">83</xref>]. Over a dozen states have enacted laws designed to ensure safeguards and guardrails for the use of AI. For example, New York State&#x2019;s RAISE Act outlines key AI risks including cybersecurity threats, system failures, criminal contact, weaponization, and critical harm resulting in death or serious injury. For more information, please see the RAISE Act (S.53/A.6453) [<xref ref-type="bibr" rid="ref84">84</xref>]. The act also has reporting requirements for managing risks including safety protocols, third-party reporting requirements, and auditing procedures. California has also enacted a similar law &#x201C;Transparency in Frontier Artificial Intelligence Act (SB53)&#x201D; that is designed to enhance online safety and install guardrails on the development of frontier AI models. Other governments are rapidly developing new laws or regulations or adopting targeted restrictions [<xref ref-type="bibr" rid="ref77">77</xref>].</p><p>Companies are expected to release successively more powerful and capable LLM-based systems in the coming months, which may introduce new benefits but also new regulatory challenges. While these laws and regulations provide general guidelines for the use of AI systems, one must keep them in mind when developing LLM-based systems targeted at underserved populations.</p></sec><sec id="s12"><title>What Are Recommendations to Improve Applicability and Trust of LLMs to Vulnerable Populations?</title><p>While there are many reporting guidelines for the use of AI tools in medicine [<xref ref-type="bibr" rid="ref85">85</xref>] and many relevant consensus statements [<xref ref-type="bibr" rid="ref86">86</xref>], it is imperative that the adoption of AI systems is based on robust, reproducible evidence obtained via scientifically sound evaluation processes.</p><p>When introducing new technologies to people with OUD, recommendations to build patient-doctor trust include stigma minimization, promotion of empathy in the patient-provider relationship, and engaging community organizations to build bridges with health care providers and institutions. With regard to LLMs, to guard against bias and hallucinations, rigorous benchmarks must evaluate LLM-based systems using medical data and tasks, so that reproducibility can be evaluated and developmental progress tracked [<xref ref-type="bibr" rid="ref29">29</xref>]. These systems must also be evaluated against, for example, frameworks that have emphasized ethical aspects, such as respecting human values and being inclusive [<xref ref-type="bibr" rid="ref87">87</xref>] and promoting transparency in AI use [<xref ref-type="bibr" rid="ref88">88</xref>,<xref ref-type="bibr" rid="ref89">89</xref>]. Humans should have ultimate oversight to ensure trustworthy AI outputs.</p><p>The unknown nature of some AI models, as the fundamental &#x201C;black box&#x201D; design of contemporary LLMs [<xref ref-type="bibr" rid="ref90">90</xref>], can promote distrust among providers [<xref ref-type="bibr" rid="ref33">33</xref>]. In the development of these tools, clinical safety, effectiveness, and health equity must be top priorities. A continuous improvement feedback mechanism should be used to promote improvements in AI tools and to report adverse events resulting from the use of AI. LLM-based systems may be able to assess for potential biases and inequity in OUD treatment plans across different patient demographics (race, ethnicity, and sex). These systems may also have a potential future role in promoting equitable OUD management decisions [<xref ref-type="bibr" rid="ref91">91</xref>].</p></sec><sec id="s13"><title>Considerations for Deployment of AI Systems for Underserved Populations</title><p>Policies toward LLM-based system implementation and deployment in health care settings that care for underserved populations have several considerations (<xref ref-type="table" rid="table6">Table 6</xref>). Consideration of the domains of engagement as well as the different stakeholders is very important. Particularly important are governmental agencies that can encourage adoption of LLMs through financial incentives or reimbursement for their use. Correspondingly, they also have a role in ensuring security, confidentiality, and privacy of LLM-based systems.</p><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Recommendations for implementation of large language model (LLM)&#x2013;based systems for underserved populations by stakeholder position.</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Domains of engagement</td><td align="left" valign="bottom" colspan="3">Stakeholders</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">Opioid treatment program (patients, leadership, administrators, and staff)</td><td align="left" valign="bottom">Academia and industry</td><td align="left" valign="bottom">Government</td></tr></thead><tbody><tr><td align="left" valign="top">Patient care workflows</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>OTP<sup><xref ref-type="table-fn" rid="table6fn1">a</xref></sup> staff explore the effectiveness of LLMs as tools to promote patient engagement.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Promoting research on areas where LLMs are effective in improving patient health outcomes.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Evaluating aspects of the OASAS<sup><xref ref-type="table-fn" rid="table6fn2">b</xref></sup> network where LLMs can improve patient health outcomes.</p></list-item></list></td></tr><tr><td align="left" valign="top">Organizational design</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>OTP staff participate in education on the use and dangers of LLMs.</p></list-item><list-item><p>OTP staff evaluate and provide feedback on LLM deployment.</p></list-item><list-item><p>OTP staff evaluate whether LLMs can safely reduce staff loads in OTP workflows.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Understanding OTP culture and community to avoid potential disruptive effects from the uptake of LLMs.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Identify areas where LLMs can safely and effectively assist within OTP workflows.</p></list-item><list-item><p>Potentially provide incentives or reimbursement to promote safe LLM use.</p></list-item></list></td></tr><tr><td align="left" valign="top">Policy</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>OTP leadership establishes clear guidance on the safe use of LLMs in patient care.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Collaborate to explore how LLM deployment can enhance patient outcomes.</p></list-item></list></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Responsible use of LLMs that is, focus on security, confidentiality, and privacy.</p></list-item></list></td></tr></tbody></table><table-wrap-foot><fn id="table6fn1"><p><sup>a</sup>OTP: Opioid Treatment Program.</p></fn><fn id="table6fn2"><p><sup>b</sup>OASAS: Office of Addiction Services and Supports.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s14"><title>Conclusions and Future Research Directions</title><p>Technological innovation is a facet of the human condition and has occurred ever since the evolution of humanity. Consideration of how to increase trust and respect in technology is vital, universal, and extremely important, especially when revolutionary technologies, such as AI, are considered.</p><p>In this paper, we discuss the adoption of new technology through the lens of DOI theory and extend the CREATE engagement framework to facilitate the implementation of LLM-based systems into OTP workflows. Future research should seek to evaluate the effectiveness of the CREATE engagement framework as related to LLM deployment. Another area for future investigation is evaluation of implementation outcomes, as would conventionally be assessed using standard implementation frameworks, such as Reach, Effectiveness, Adoption, Implementation, and Maintenance (RE-AIM) [<xref ref-type="bibr" rid="ref92">92</xref>] or Consolidated Framework for Implementation Research (CFIR) [<xref ref-type="bibr" rid="ref93">93</xref>]. Evaluation matrices for each of the 6 components of the CREATE framework are listed in the <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref> and could serve to highlight specific areas for future research inquiry.</p><p>Advances in foundation models, such as multimodal processing, increased planning capabilities, and subsequently the development of autonomous LLM agents with the ability to interact with a wide array of tools, emphasize the need for the CREATE framework even further. &#x201C;Advancement&#x201D; drives the adoption of these newer innovations into various aspects of patient care, while &#x201C;trust&#x201D; ensures that it is done under rigorous oversight by giving patient safety and confidentiality the highest priority. At the same time, these advances have led to the rise of new challenges, underscoring that &#x201C;expertise&#x201D; continues to incorporate the needed skills as new innovations are introduced in LLM-based systems. These developments highlight the indispensable role of human experts in ensuring that LLM-based systems are safely and effectively integrated into clinical workflows.</p><p>Olawade et al [<xref ref-type="bibr" rid="ref94">94</xref>] present a narrative review of HITL AI in health care and discuss associated implementation challenges and future research directions. Along with HITL, further research is needed on human-computer interaction involving AI tools. These tools must be developed in a manner that facilitates the collaboration between humans and AI to build trust, ensure usability, and promote adoption [<xref ref-type="bibr" rid="ref95">95</xref>]. Additionally, an important research area is understanding the economic implications of deploying these tools in health care [<xref ref-type="bibr" rid="ref96">96</xref>].</p><p>The multifaceted pyramid of data, evidence generation, and risk associated with LLM-based systems illustrates the relationships between the different faces and associated levels of evidence generation. The information present in the collected data is fundamental for high-quality evidence generation. The progression among the different levels of the data face of the pyramid can be interpreted as higher levels in the data hierarchy indicate higher information content, which, subsequently, increases the risk associated with the AI system deployment, while simultaneously increasing the quality of evidence associated with the health care recommendations provided to patients. How to formalize these relationships and their interaction with the CREATE framework remains the topic of a different paper.</p><p>While the science behind AI is relatively new, issues to resolve remain and must be satisfactorily addressed to ensure the public&#x2019;s acceptability of the technology. Recently enacted legislation at the state level to provide guardrails for the use of AI is a step in the right direction, although additional topics remain when considering targeting underserved populations. Multidisciplinary perspectives represented by linguistics, computer science, statistical sciences, and medicine are needed to optimize the health care application of LLMs.</p></sec></body><back><ack><p>The authors acknowledge the assistance of Mr. Kenneth Bossert, Former Administrator, Drug Abuse, Research, and Treatment Center (DART) Opioid Treatment Program (OTP), Buffalo, New York. They also acknowledge Dr Ashly Jordan, Office of Addiction Services and Supports (OASAS), and Mr Andrew Heck, MPH, OASAS, for helpful discussions.</p></ack><notes><sec><title>Funding</title><p>This work was supported by Patient-Centered Outcomes Research Institute (PCORI) Award ME-2024C1-37584 [MM] [<xref ref-type="bibr" rid="ref97">97</xref>], and partially supported by the Troup Fund of the Kaleida Health Foundation [MM]. The statements in this work are solely the responsibility of the authors and do not necessarily represent the views of PCORI, its Board of Governors or Methodology Committee. Study funders were not involved in data collection, analysis, or manuscript preparation.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization, Funding acquisition, Investigation, Methodology, Project administration, Resources, Supervision, Validation, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing: MM</p><p>Investigation, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing: RM</p><p>Investigation, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing: JG, YT, AD</p><p>Writing &#x2013; review &#x0026; editing: LB</p><p>Conceptualization, Investigation, Methodology, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing: AHT</p></fn><fn fn-type="conflict"><p>AHT has received research support from Gilead Sciences, Novo Nordisk, AstraZeneca, and Salix paid to his institution. AHT has received consultant and honoraria support from AbbVie, Gilead, Novo Nordisk, and Madrigal. AHT is also President of Empath Medical Innovations. MM is Vice President of Empath Medical Innovations.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AMIA</term><def><p>American Medical Informatics Association</p></def></def-item><def-item><term id="abb2">CFIR</term><def><p>Consolidated Framework for Implementation Research</p></def></def-item><def-item><term id="abb3">CREATE</term><def><p>Culture, Respect, Education, Advancement, Trust, and Expertise</p></def></def-item><def-item><term id="abb4">DOI</term><def><p>diffusion of innovations</p></def></def-item><def-item><term id="abb5">EHR</term><def><p>electronic health record</p></def></def-item><def-item><term id="abb6">EU</term><def><p>European Union</p></def></def-item><def-item><term id="abb7">GenAI</term><def><p>generative AI</p></def></def-item><def-item><term id="abb8">HIPAA</term><def><p>Health Insurance Portability and Accountability Act</p></def></def-item><def-item><term id="abb9">HIT</term><def><p>health information technology</p></def></def-item><def-item><term id="abb10">HITL</term><def><p>human-in-the-loop</p></def></def-item><def-item><term id="abb11">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb12">NLP</term><def><p>natural language processing</p></def></def-item><def-item><term id="abb13">OASAS</term><def><p>Office of Addiction Services and Supports</p></def></def-item><def-item><term id="abb14">OTP</term><def><p>Opioid Treatment Program</p></def></def-item><def-item><term id="abb15">OUD</term><def><p>opioid use disorder</p></def></def-item><def-item><term id="abb16">PCORI</term><def><p>Patient-Centered Outcomes Research Institute</p></def></def-item><def-item><term id="abb17">RE-AIM</term><def><p>Reach, Effectiveness, Adoption, Implementation, and Maintenance</p></def></def-item><def-item><term id="abb18">STS</term><def><p>socio-technical system</p></def></def-item><def-item><term id="abb19">WHO</term><def><p>World Health Organization</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Angus</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Khera</surname><given-names>R</given-names> </name><name name-style="western"><surname>Lieu</surname><given-names>T</given-names> </name><etal/></person-group><article-title>AI, health, and health care today and tomorrow: the JAMA summit report on artificial intelligence</article-title><source>JAMA</source><year>2025</year><month>11</month><day>11</day><volume>334</volume><issue>18</issue><fpage>1650</fpage><lpage>1664</lpage><pub-id pub-id-type="doi">10.1001/jama.2025.18490</pub-id><pub-id pub-id-type="medline">41082366</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Papalamprakopoulou</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Roussos</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ntagianta</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Considerations for equitable distribution of digital healthcare for people who use drugs</article-title><source>BMC Health Serv Res</source><year>2025</year><month>04</month><day>10</day><volume>25</volume><issue>1</issue><fpage>531</fpage><pub-id pub-id-type="doi">10.1186/s12913-025-12619-7</pub-id><pub-id pub-id-type="medline">40211324</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Talal</surname><given-names>AH</given-names> </name><name name-style="western"><surname>George</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Talal</surname><given-names>LA</given-names> </name><etal/></person-group><article-title>Engaging people who use drugs in clinical research: integrating facilitated telemedicine for HCV into substance use treatment</article-title><source>Res Involv Engagem</source><year>2023</year><month>08</month><day>2</day><volume>9</volume><issue>1</issue><fpage>63</fpage><pub-id pub-id-type="doi">10.1186/s40900-023-00474-x</pub-id><pub-id pub-id-type="medline">37533127</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Talal</surname><given-names>AH</given-names> </name><name name-style="western"><surname>Jaanim&#x00E4;gi</surname><given-names>U</given-names> </name><name name-style="western"><surname>Davis</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Facilitating engagement of persons with opioid use disorder in treatment for hepatitis C virus infection via telemedicine: stories of onsite case managers</article-title><source>J Subst Abuse Treat</source><year>2021</year><month>08</month><volume>127</volume><fpage>108421</fpage><pub-id pub-id-type="doi">10.1016/j.jsat.2021.108421</pub-id><pub-id pub-id-type="medline">34134875</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dole</surname><given-names>VP</given-names> </name><name name-style="western"><surname>Nyswander</surname><given-names>ME</given-names> </name><name name-style="western"><surname>Kreek</surname><given-names>MJ</given-names> </name></person-group><article-title>Narcotic blockade</article-title><source>Arch Intern Med</source><year>1966</year><month>10</month><volume>118</volume><issue>4</issue><fpage>304</fpage><lpage>309</lpage><pub-id pub-id-type="medline">4162686</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Califf</surname><given-names>RM</given-names> </name><name name-style="western"><surname>Robb</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Bindman</surname><given-names>AB</given-names> </name><etal/></person-group><article-title>Transforming evidence generation to support health and health care decisions</article-title><source>N Engl J Med</source><year>2016</year><month>12</month><day>15</day><volume>375</volume><issue>24</issue><fpage>2395</fpage><lpage>2400</lpage><pub-id pub-id-type="doi">10.1056/NEJMsb1610128</pub-id><pub-id pub-id-type="medline">27974039</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jackson</surname><given-names>GP</given-names> </name><name name-style="western"><surname>Shortliffe</surname><given-names>EH</given-names> </name></person-group><article-title>Understanding the evidence for artificial intelligence in healthcare</article-title><source>BMJ Qual Saf</source><year>2025</year><month>06</month><day>19</day><volume>34</volume><issue>7</issue><fpage>421</fpage><lpage>424</lpage><pub-id pub-id-type="doi">10.1136/bmjqs-2025-018559</pub-id><pub-id pub-id-type="medline">40246317</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Rogers</surname><given-names>EM</given-names> </name></person-group><source>Diffusion of Innovations</source><year>2003</year><edition>5</edition><publisher-name>Free Press</publisher-name><pub-id pub-id-type="other">9780743258234</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="web"><article-title>Generative artificial intelligence: overview, issues, and considerations for congress</article-title><source>United States Congress</source><year>2025</year><access-date>2026-09-14</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.congress.gov/crs_external_products/IF/HTML/IF12426.html">https://www.congress.gov/crs_external_products/IF/HTML/IF12426.html</ext-link></comment></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Vaswani</surname><given-names>A</given-names> </name><name name-style="western"><surname>Shazeer</surname><given-names>N</given-names> </name><name name-style="western"><surname>Parmar</surname><given-names>N</given-names> </name><name name-style="western"><surname>Uszkoreit</surname><given-names>J</given-names> </name><name name-style="western"><surname>Jones</surname><given-names>L</given-names> </name><name name-style="western"><surname>Gomez</surname><given-names>AN</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Guyon</surname><given-names>I</given-names> </name><name name-style="western"><surname>Luxburg</surname><given-names>UV</given-names> </name><name name-style="western"><surname>Bengio</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wallach</surname><given-names>H</given-names> </name><name name-style="western"><surname>Fergus</surname><given-names>R</given-names> </name><name name-style="western"><surname>Vishwanathan</surname><given-names>S</given-names> </name></person-group><article-title>Attention is all you need</article-title><year>2017</year><conf-name>NIPS&#x2019;17: Proceedings of the 31st International Conference on Neural Information Processing Systems</conf-name><conf-date>Dec 4-9, 2017</conf-date><pub-id pub-id-type="doi">10.5555/3295222.3295349</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aydin</surname><given-names>S</given-names> </name><name name-style="western"><surname>Karabacak</surname><given-names>M</given-names> </name><name name-style="western"><surname>Vlachos</surname><given-names>V</given-names> </name><name name-style="western"><surname>Margetis</surname><given-names>K</given-names> </name></person-group><article-title>Large language models in patient education: a scoping review of applications in medicine</article-title><source>Front Med (Lausanne)</source><year>2024</year><volume>11</volume><fpage>1477898</fpage><pub-id pub-id-type="doi">10.3389/fmed.2024.1477898</pub-id><pub-id pub-id-type="medline">39534227</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Generative AI in medical education: feasibility and educational value of LLM-generated clinical cases with MCQs</article-title><source>BMC Med Educ</source><year>2025</year><month>10</month><day>27</day><volume>25</volume><issue>1</issue><fpage>1502</fpage><pub-id pub-id-type="doi">10.1186/s12909-025-08085-8</pub-id><pub-id pub-id-type="medline">41146115</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wilhelm</surname><given-names>TI</given-names> </name><name name-style="western"><surname>Roos</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kaczmarczyk</surname><given-names>R</given-names> </name></person-group><article-title>Large language models for therapy recommendations across 3 clinical specialties: comparative study</article-title><source>J Med Internet Res</source><year>2023</year><month>10</month><day>30</day><volume>25</volume><fpage>e49324</fpage><pub-id pub-id-type="doi">10.2196/49324</pub-id><pub-id pub-id-type="medline">37902826</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Usuyama</surname><given-names>N</given-names> </name><name name-style="western"><surname>Wong</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Naumann</surname><given-names>T</given-names> </name><name name-style="western"><surname>Poon</surname><given-names>H</given-names> </name></person-group><article-title>Biomedical natural language processing in the era of large language models</article-title><source>Annu Rev Biomed Data Sci</source><year>2025</year><month>08</month><volume>8</volume><issue>1</issue><fpage>471</fpage><lpage>490</lpage><pub-id pub-id-type="doi">10.1146/annurev-biodatasci-103123-095406</pub-id><pub-id pub-id-type="medline">40245359</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhou</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>AZ</given-names> </name><name name-style="western"><surname>Shah</surname><given-names>D</given-names> </name><name name-style="western"><surname>Schwab-Reese</surname><given-names>LM</given-names> </name><name name-style="western"><surname>DE Choudhury</surname><given-names>M</given-names> </name></person-group><article-title>A risk taxonomy and reflection tool for large language model adoption in public health</article-title><source>Proc ACM Hum Comput Interact</source><year>2025</year><month>11</month><volume>9</volume><issue>7</issue><fpage>CSCW363</fpage><pub-id pub-id-type="doi">10.1145/3757544</pub-id><pub-id pub-id-type="medline">41716439</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="web"><article-title>Sociotechnical systems</article-title><source>Taylor &#x0026; Francis Knowledge Centers</source><access-date>2026-09-14</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://taylorandfrancis.com/knowledge/Engineering_and_technology/Computer_science/Sociotechnical_systems">https://taylorandfrancis.com/knowledge/Engineering_and_technology/Computer_science/Sociotechnical_systems</ext-link></comment></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sittig</surname><given-names>DF</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>H</given-names> </name></person-group><article-title>A new sociotechnical model for studying health information technology in complex adaptive healthcare systems</article-title><source>Qual Saf Health Care</source><year>2010</year><month>10</month><volume>19 Suppl 3</volume><issue>Suppl 3</issue><fpage>i68</fpage><lpage>74</lpage><pub-id pub-id-type="doi">10.1136/qshc.2010.042085</pub-id><pub-id pub-id-type="medline">20959322</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Selbst</surname><given-names>AD</given-names> </name><name name-style="western"><surname>Boyd</surname><given-names>D</given-names> </name><name name-style="western"><surname>Friedler</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Venkatasubramanian</surname><given-names>S</given-names> </name><name name-style="western"><surname>Vertesi</surname><given-names>J</given-names> </name></person-group><article-title>Fairness and abstraction in sociotechnical systems</article-title><year>2019</year><month>01</month><day>29</day><conf-name>FAT* &#x2019;19: Proceedings of the Conference on Fairness, Accountability, and Transparency</conf-name><conf-date>Jan 29-31, 2019</conf-date><conf-loc>Atlanta GA USA</conf-loc><fpage>59</fpage><lpage>68</lpage><pub-id pub-id-type="doi">10.1145/3287560.3287598</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="web"><article-title>Artificial intelligence</article-title><source>Epic systems c</source><access-date>2026-09-14</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.epic.com/software/ai">https://www.epic.com/software/ai</ext-link></comment></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>SI</given-names> </name><name name-style="western"><surname>Park</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Enhancing patient participation in emergency department through patient-friendly clinical notes generated by large language models</article-title><source>Sci Rep</source><year>2025</year><month>12</month><day>5</day><volume>16</volume><issue>1</issue><fpage>1409</fpage><pub-id pub-id-type="doi">10.1038/s41598-025-31113-y</pub-id><pub-id pub-id-type="medline">41350380</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>G</given-names> </name><etal/></person-group><article-title>A generalist medical language model for disease diagnosis assistance</article-title><source>Nat Med</source><year>2025</year><month>03</month><volume>31</volume><issue>3</issue><fpage>932</fpage><lpage>942</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03416-6</pub-id><pub-id pub-id-type="medline">39779927</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Panagoulias</surname><given-names>DP</given-names> </name><name name-style="western"><surname>Virvou</surname><given-names>M</given-names> </name><name name-style="western"><surname>Tsihrintzis</surname><given-names>GA</given-names> </name></person-group><article-title>Evaluating LLM--generated multimodal diagnosis from medical images and symptom analysis</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 28, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2402.01730</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Panagoulias</surname><given-names>DP</given-names> </name><name name-style="western"><surname>Palamidas</surname><given-names>FA</given-names> </name><name name-style="western"><surname>Virvou</surname><given-names>M</given-names> </name><name name-style="western"><surname>Tsihrintzis</surname><given-names>GA</given-names> </name></person-group><article-title>Evaluating the potential of llms and chatgpt on medical diagnosis and treatment</article-title><conf-name>2023 14th International Conference on Information, Intelligence, Systems &#x0026; Applications (IISA)</conf-name><conf-date>Jul 10-12, 2023</conf-date><pub-id pub-id-type="doi">10.1109/IISA59645.2023.10345968</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Levra</surname><given-names>AG</given-names> </name><name name-style="western"><surname>Gatti</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mene</surname><given-names>R</given-names> </name><etal/></person-group><article-title>A large language model-based clinical decision support system for syncope recognition in the emergency department: a framework for clinical workflow integration</article-title><source>Eur J Intern Med</source><year>2025</year><month>01</month><volume>131</volume><fpage>113</fpage><lpage>120</lpage><pub-id pub-id-type="doi">10.1016/j.ejim.2024.09.017</pub-id><pub-id pub-id-type="medline">39341748</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="web"><article-title>Tools and resources to support engagement in research</article-title><source>Patient-Centered Outcomes Research Institute (PCORI)</source><access-date>2026-09-24</access-date><comment><ext-link ext-link-type="uri" xlink:href="http://pcori.org/engagement-research/tools-and-resources-support-engagement-research">http://pcori.org/engagement-research/tools-and-resources-support-engagement-research</ext-link></comment></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sarraju</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bruemmer</surname><given-names>D</given-names> </name><name name-style="western"><surname>Van Iterson</surname><given-names>E</given-names> </name><name name-style="western"><surname>Cho</surname><given-names>L</given-names> </name><name name-style="western"><surname>Rodriguez</surname><given-names>F</given-names> </name><name name-style="western"><surname>Laffin</surname><given-names>L</given-names> </name></person-group><article-title>Appropriateness of cardiovascular disease prevention recommendations obtained from a popular online chat-based artificial intelligence model</article-title><source>JAMA</source><year>2023</year><month>03</month><day>14</day><volume>329</volume><issue>10</issue><fpage>842</fpage><lpage>844</lpage><pub-id pub-id-type="doi">10.1001/jama.2023.1044</pub-id><pub-id pub-id-type="medline">36735264</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mika</surname><given-names>AP</given-names> </name><name name-style="western"><surname>Martin</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Engstrom</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Polkowski</surname><given-names>GG</given-names> </name><name name-style="western"><surname>Wilson</surname><given-names>JM</given-names> </name></person-group><article-title>Assessing ChatGPT responses to common patient questions regarding total hip arthroplasty</article-title><source>J Bone Joint Surg Am</source><year>2023</year><month>10</month><day>4</day><volume>105</volume><issue>19</issue><fpage>1519</fpage><lpage>1526</lpage><pub-id pub-id-type="doi">10.2106/JBJS.23.00209</pub-id><pub-id pub-id-type="medline">37459402</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Grewal</surname><given-names>H</given-names> </name><name name-style="western"><surname>Dhillon</surname><given-names>G</given-names> </name><name name-style="western"><surname>Monga</surname><given-names>V</given-names> </name><etal/></person-group><article-title>Radiology gets chatty: the ChatGPT saga unfolds</article-title><source>Cureus</source><year>2023</year><month>06</month><volume>15</volume><issue>6</issue><fpage>e40135</fpage><pub-id pub-id-type="doi">10.7759/cureus.40135</pub-id><pub-id pub-id-type="medline">37425598</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Omiye</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Gui</surname><given-names>H</given-names> </name><name name-style="western"><surname>Rezaei</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Zou</surname><given-names>J</given-names> </name><name name-style="western"><surname>Daneshjou</surname><given-names>R</given-names> </name></person-group><article-title>Large language models in medicine: the potentials and pitfalls: a narrative review</article-title><source>Ann Intern Med</source><year>2024</year><month>02</month><volume>177</volume><issue>2</issue><fpage>210</fpage><lpage>220</lpage><pub-id pub-id-type="doi">10.7326/M23-2772</pub-id><pub-id pub-id-type="medline">38285984</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>S</given-names> </name><name name-style="western"><surname>McCoy</surname><given-names>AB</given-names> </name><name name-style="western"><surname>Wright</surname><given-names>AP</given-names> </name><etal/></person-group><article-title>Leveraging large language models for generating responses to patient messages</article-title><source>medRxiv</source><year>2023</year><month>07</month><day>16</day><fpage>2023.07.14.23292669</fpage><pub-id pub-id-type="doi">10.1101/2023.07.14.23292669</pub-id><pub-id pub-id-type="medline">37503263</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Djalilian</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ali</surname><given-names>MJ</given-names> </name></person-group><article-title>ChatGPT and ophthalmology: exploring its potential with discharge summaries and operative notes</article-title><source>Semin Ophthalmol</source><year>2023</year><month>07</month><volume>38</volume><issue>5</issue><fpage>503</fpage><lpage>507</lpage><pub-id pub-id-type="doi">10.1080/08820538.2023.2209166</pub-id><pub-id pub-id-type="medline">37133418</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sorin</surname><given-names>V</given-names> </name><name name-style="western"><surname>Klang</surname><given-names>E</given-names> </name><name name-style="western"><surname>Sklair-Levy</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Large language model (ChatGPT) as a support tool for breast tumor board</article-title><source>NPJ Breast Cancer</source><year>2023</year><month>05</month><day>30</day><volume>9</volume><issue>1</issue><fpage>44</fpage><pub-id pub-id-type="doi">10.1038/s41523-023-00557-8</pub-id><pub-id pub-id-type="medline">37253791</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Daneshvar</surname><given-names>N</given-names> </name><name name-style="western"><surname>Pandita</surname><given-names>D</given-names> </name><name name-style="western"><surname>Erickson</surname><given-names>S</given-names> </name><name name-style="western"><surname>Snyder Sulmasy</surname><given-names>L</given-names> </name><name name-style="western"><surname>DeCamp</surname><given-names>M</given-names> </name><name name-style="western"><surname>Committee</surname><given-names>A</given-names> </name></person-group><article-title>Artificial Intelligence in the provision of health care: an American College of Physicians policy position paper</article-title><source>Ann Intern Med</source><year>2024</year><month>07</month><volume>177</volume><issue>7</issue><fpage>964</fpage><lpage>967</lpage><pub-id pub-id-type="doi">10.7326/M24-0146</pub-id><pub-id pub-id-type="medline">38830215</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sidorov</surname><given-names>G</given-names> </name><name name-style="western"><surname>Ahmad</surname><given-names>M</given-names> </name><name name-style="western"><surname>Basile</surname><given-names>P</given-names> </name><name name-style="western"><surname>Waqas</surname><given-names>M</given-names> </name><name name-style="western"><surname>Orji</surname><given-names>R</given-names> </name><name name-style="western"><surname>Batyrshin</surname><given-names>I</given-names> </name></person-group><article-title>Monitoring opioid-related social media chatter using natural language processing and large language models: temporal analysis</article-title><source>JMIR Infodemiology</source><year>2025</year><month>11</month><day>4</day><volume>5</volume><fpage>e77279</fpage><pub-id pub-id-type="doi">10.2196/77279</pub-id><pub-id pub-id-type="medline">41187282</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sousa-Pinto</surname><given-names>B</given-names> </name><name name-style="western"><surname>Marques-Cruz</surname><given-names>M</given-names> </name><name name-style="western"><surname>Neumann</surname><given-names>I</given-names> </name><etal/></person-group><article-title>Guidelines international network: principles for use of artificial intelligence in the health guideline enterprise</article-title><source>Ann Intern Med</source><year>2025</year><month>03</month><volume>178</volume><issue>3</issue><fpage>408</fpage><lpage>415</lpage><pub-id pub-id-type="doi">10.7326/ANNALS-24-02338</pub-id><pub-id pub-id-type="medline">39869912</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cao</surname><given-names>C</given-names> </name><name name-style="western"><surname>Sang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Arora</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Development of prompt templates for large language model-driven screening in systematic reviews</article-title><source>Ann Intern Med</source><year>2025</year><month>03</month><volume>178</volume><issue>3</issue><fpage>389</fpage><lpage>401</lpage><pub-id pub-id-type="doi">10.7326/ANNALS-24-02189</pub-id><pub-id pub-id-type="medline">39993313</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>McNemar</surname><given-names>E</given-names> </name></person-group><article-title>How does artificial intelligence compare to augmented intelligence</article-title><source>TechTarget Health IT Analytics</source><year>2021</year><access-date>2026-09-14</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.techtarget.com/healthtechanalytics/news/366591056/How-Does-Artificial-Intelligence-Compare-to-Augmented-Intelligence">https://www.techtarget.com/healthtechanalytics/news/366591056/How-Does-Artificial-Intelligence-Compare-to-Augmented-Intelligence</ext-link></comment></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="web"><article-title>Augmented intelligence in medicine</article-title><source>American Medical Association</source><access-date>2026-09-14</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.ama-assn.org/practice-management/digital-health/augmented-intelligence-medicine">https://www.ama-assn.org/practice-management/digital-health/augmented-intelligence-medicine</ext-link></comment></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>P</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>X</given-names> </name><name name-style="western"><surname>Cheng</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Human-in-the-loop interactive report generation for chronic disease adherence</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 10, 2026</comment><pub-id pub-id-type="doi">10.48550/arXiv.2601.06364</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Kieseberg</surname><given-names>P</given-names> </name><name name-style="western"><surname>Weippl</surname><given-names>E</given-names> </name><name name-style="western"><surname>Holzinger</surname><given-names>A</given-names> </name></person-group><article-title>Trust for the &#x201C;doctor in the loop&#x201D;</article-title><source>ERCIM News</source><year>2016</year><month>01</month><day>13</day><access-date>2026-09-24</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://ercim-news.ercim.eu/en104/special/trust-for-the-doctor-in-the-loop">https://ercim-news.ercim.eu/en104/special/trust-for-the-doctor-in-the-loop</ext-link></comment></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Caragliano</surname><given-names>AN</given-names> </name><name name-style="western"><surname>Ruffini</surname><given-names>F</given-names> </name><name name-style="western"><surname>Greco</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Doctor-in-the-Loop: an explainable, multi-view deep learning framework for predicting pathological response in non-small cell lung cancer</article-title><source>Image Vis Comput</source><year>2025</year><month>09</month><volume>161</volume><fpage>105630</fpage><pub-id pub-id-type="doi">10.1016/j.imavis.2025.105630</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liao</surname><given-names>QV</given-names> </name><name name-style="western"><surname>Wortman Vaughan</surname><given-names>J</given-names> </name></person-group><article-title>AI transparency in the age of LLMs: a human-centered research roadmap</article-title><source>Harvard Data Science Review</source><year>2024</year><issue>Special Issue 5</issue><pub-id pub-id-type="doi">10.1162/99608f92.8036d03b</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="book"><person-group person-group-type="editor"><name name-style="western"><surname>Bouzoubaa</surname><given-names>L</given-names> </name><name name-style="western"><surname>Aghakhani</surname><given-names>E</given-names> </name><name name-style="western"><surname>Rezapour</surname><given-names>R</given-names> </name></person-group><source>Words Matter: Reducing Stigma in Online Conversations about Substance Use with Large Language Models</source><year>2024</year><publisher-name>Association for Computational Linguistics</publisher-name><pub-id pub-id-type="doi">10.18653/v1/2024.emnlp-main.516</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lo Bianco</surname><given-names>G</given-names> </name><name name-style="western"><surname>Robinson</surname><given-names>CL</given-names> </name><name name-style="western"><surname>D&#x2019;Angelo</surname><given-names>FP</given-names> </name><etal/></person-group><article-title>Effectiveness of generative artificial intelligence-driven responses to patient concerns in long-term opioid therapy: cross-model assessment</article-title><source>Biomedicines</source><year>2025</year><month>03</month><day>5</day><volume>13</volume><issue>3</issue><fpage>636</fpage><pub-id pub-id-type="doi">10.3390/biomedicines13030636</pub-id><pub-id pub-id-type="medline">40149612</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nateghi Haredasht</surname><given-names>F</given-names> </name><name name-style="western"><surname>Lopez</surname><given-names>I</given-names> </name><name name-style="western"><surname>Tate</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Predicting treatment retention in medication for opioid use disorder: a machine learning approach using NLP and LLM-derived clinical features</article-title><source>J Am Med Inform Assoc</source><year>2025</year><month>12</month><day>1</day><volume>32</volume><issue>12</issue><fpage>1865</fpage><lpage>1876</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocaf157</pub-id><pub-id pub-id-type="medline">40977375</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Voigt</surname><given-names>H</given-names> </name><name name-style="western"><surname>Sugamiya</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Lawonn</surname><given-names>K</given-names> </name><name name-style="western"><surname>Zarrie&#x00DF;</surname><given-names>S</given-names> </name><name name-style="western"><surname>Takanishi</surname><given-names>A</given-names> </name></person-group><article-title>LLM-powered virtual patient agents for interactive clinical skills training with automated feedback</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 19, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2508.13943</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Agweyu</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mwaniki</surname><given-names>P</given-names> </name><name name-style="western"><surname>Menon</surname><given-names>V</given-names> </name><etal/></person-group><article-title>Generative AI-enabled clinical decision support system in primary care: a pragmatic, cluster-randomized trial</article-title><source>Nat Med</source><year>2026</year><month>08</month><volume>32</volume><issue>8</issue><fpage>3032</fpage><lpage>3039</lpage><pub-id pub-id-type="doi">10.1038/s41591-026-04503-6</pub-id><pub-id pub-id-type="medline">42362867</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ong</surname><given-names>JCL</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>L</given-names> </name><name name-style="western"><surname>Elangovan</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Large language model as clinical decision support system augments medication safety in 16 clinical specialties</article-title><source>Cell Rep Med</source><year>2025</year><month>10</month><day>21</day><volume>6</volume><issue>10</issue><fpage>102323</fpage><pub-id pub-id-type="doi">10.1016/j.xcrm.2025.102323</pub-id><pub-id pub-id-type="medline">40997804</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sethi</surname><given-names>R</given-names> </name><name name-style="western"><surname>Caskey</surname><given-names>J</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Detecting stigmatizing language in clinical notes with large language models for addiction care</article-title><source>medRxiv</source><year>2025</year><month>08</month><day>12</day><fpage>2025.08.08.25333315</fpage><pub-id pub-id-type="doi">10.1101/2025.08.08.25333315</pub-id><pub-id pub-id-type="medline">40832420</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kaiyrbekov</surname><given-names>K</given-names> </name><name name-style="western"><surname>Dobbins</surname><given-names>NJ</given-names> </name><name name-style="western"><surname>Mooney</surname><given-names>SD</given-names> </name></person-group><article-title>Automated survey collection with LLM-based conversational agents</article-title><source>JAMIA Open</source><year>2025</year><month>10</month><volume>8</volume><issue>5</issue><fpage>ooaf103</fpage><pub-id pub-id-type="doi">10.1093/jamiaopen/ooaf103</pub-id><pub-id pub-id-type="medline">41180893</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Van Veen</surname><given-names>D</given-names> </name><name name-style="western"><surname>Van Uden</surname><given-names>C</given-names> </name><name name-style="western"><surname>Blankemeier</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Adapted large language models can outperform medical experts in clinical text summarization</article-title><source>Nat Med</source><year>2024</year><month>04</month><volume>30</volume><issue>4</issue><fpage>1134</fpage><lpage>1142</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-02855-5</pub-id><pub-id pub-id-type="medline">38413730</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Song</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Jang</surname><given-names>JY</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>H</given-names> </name><name name-style="western"><surname>Ko</surname><given-names>YG</given-names> </name><name name-style="western"><surname>You</surname><given-names>SC</given-names> </name></person-group><article-title>Transforming free-text coronary angiography reports into structured, analyzable data using large language models</article-title><source>Sci Rep</source><year>2026</year><month>01</month><day>3</day><volume>16</volume><issue>1</issue><fpage>2360</fpage><pub-id pub-id-type="doi">10.1038/s41598-025-32150-3</pub-id><pub-id pub-id-type="medline">41484419</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mansoor</surname><given-names>I</given-names> </name><name name-style="western"><surname>Mohammed</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Blythe</surname><given-names>S</given-names> </name></person-group><article-title>BPI25-012: developing an artificial intelligence tool for personalized breast cancer treatment plans based on the NCCN guidelines</article-title><source>J Natl Compr Canc Netw</source><year>2025</year><volume>23</volume><issue>3.5</issue><fpage>250215698</fpage><pub-id pub-id-type="doi">10.6004/jnccn.2024.7135</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Li</surname><given-names>M</given-names> </name><etal/></person-group><article-title>A feasibility study of automating radiotherapy planning with large language model agents</article-title><source>Phys Med Biol</source><year>2025</year><month>03</month><day>21</day><volume>70</volume><issue>7</issue><pub-id pub-id-type="doi">10.1088/1361-6560/adbff1</pub-id><pub-id pub-id-type="medline">40073507</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>He</surname><given-names>W</given-names> </name><name name-style="western"><surname>Liao</surname><given-names>J</given-names> </name><name name-style="western"><surname>Pan</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Feasibility of LLM-assisted post-discharge tuberculosis care: a comparative study of medication counseling, patient education, and follow-up planning</article-title><source>Digit HEALTH</source><year>2026</year><volume>12</volume><fpage>20552076261472569</fpage><pub-id pub-id-type="doi">10.1177/20552076261472569</pub-id><pub-id pub-id-type="medline">42571185</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Fu</surname><given-names>W</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Piao</surname><given-names>J</given-names> </name><etal/></person-group><article-title>A survey on responsible llms: inherent risk, malicious use, and mitigation strategy</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 16, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2501.09431</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jiao</surname><given-names>J</given-names> </name><name name-style="western"><surname>Afroogh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Phillips</surname><given-names>C</given-names> </name></person-group><article-title>Navigating LLM ethics: advancements, challenges, and future directions</article-title><source>AI Ethics</source><year>2025</year><month>12</month><volume>5</volume><issue>6</issue><fpage>5795</fpage><lpage>5819</lpage><pub-id pub-id-type="doi">10.1007/s43681-025-00814-5</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Bender</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Hanna</surname><given-names>A</given-names> </name></person-group><source>The AI Con: How to Fight Big Tech&#x2019;s Hype and Create the Future We Want</source><year>2025</year><publisher-name>HarperCollins</publisher-name><pub-id pub-id-type="other">9780063418554</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Amugongo</surname><given-names>LM</given-names> </name><name name-style="western"><surname>Mascheroni</surname><given-names>P</given-names> </name><name name-style="western"><surname>Brooks</surname><given-names>S</given-names> </name><name name-style="western"><surname>Doering</surname><given-names>S</given-names> </name><name name-style="western"><surname>Seidel</surname><given-names>J</given-names> </name></person-group><article-title>Retrieval augmented generation for large language models in healthcare: a systematic review</article-title><source>PLOS Digit Health</source><year>2025</year><month>06</month><volume>4</volume><issue>6</issue><fpage>e0000877</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000877</pub-id><pub-id pub-id-type="medline">40498738</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sasse</surname><given-names>K</given-names> </name><etal/></person-group><article-title>When helpfulness backfires: LLMs and the risk of false medical information due to sycophantic behavior</article-title><source>NPJ Digit Med</source><year>2025</year><month>10</month><day>17</day><volume>8</volume><issue>1</issue><fpage>605</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-02008-z</pub-id><pub-id pub-id-type="medline">41107408</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Rebedea</surname><given-names>T</given-names> </name><name name-style="western"><surname>Dinu</surname><given-names>R</given-names> </name><name name-style="western"><surname>Sreedhar</surname><given-names>MN</given-names> </name><name name-style="western"><surname>Parisien</surname><given-names>C</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Cohen</surname><given-names>J</given-names> </name></person-group><source>NeMo Guardrails: A Toolkit for Controllable and Safe LLM Applications with Programmable Rails</source><year>2023</year><publisher-name>Association for Computational Linguistics</publisher-name><pub-id pub-id-type="doi">10.18653/v1/2023.emnlp-demo.40</pub-id></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhao</surname><given-names>H</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>H</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Explainability for large language models: a survey</article-title><source>ACM Trans Intell Syst Technol</source><year>2024</year><month>04</month><day>30</day><volume>15</volume><issue>2</issue><fpage>1</fpage><lpage>38</lpage><pub-id pub-id-type="doi">10.1145/3639372</pub-id></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Yuan</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><name name-style="western"><surname>Ding</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Understanding and mitigating numerical sources of nondeterminism in LLM inference</article-title><year>2025</year><conf-name>Advances in Neural Information Processing Systems 38</conf-name><conf-date>Dec 2-7, 2025</conf-date><pub-id pub-id-type="doi">10.52202/085713-5653</pub-id></nlm-citation></ref><ref id="ref64"><label>64</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shorinwa</surname><given-names>O</given-names> </name><name name-style="western"><surname>Mei</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Lidard</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ren</surname><given-names>AZ</given-names> </name><name name-style="western"><surname>Majumdar</surname><given-names>A</given-names> </name></person-group><article-title>A survey on uncertainty quantification of large language models: taxonomy, open research challenges, and future directions</article-title><source>ACM Comput Surv</source><year>2026</year><month>02</month><day>28</day><volume>58</volume><issue>3</issue><fpage>1</fpage><lpage>38</lpage><pub-id pub-id-type="doi">10.1145/3744238</pub-id></nlm-citation></ref><ref id="ref65"><label>65</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Atf</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Safavi-Naini</surname><given-names>SAA</given-names> </name><name name-style="western"><surname>Lewis</surname><given-names>PR</given-names> </name><name name-style="western"><surname>Mahjoubfar</surname><given-names>A</given-names> </name><name name-style="western"><surname>Naderi</surname><given-names>N</given-names> </name><name name-style="western"><surname>Savage</surname><given-names>TR</given-names> </name><etal/></person-group><article-title>The challenge of uncertainty quantification of large language models in medicine</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 7, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2504.05278</pub-id></nlm-citation></ref><ref id="ref66"><label>66</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Savage</surname><given-names>T</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Gallo</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Large language model uncertainty proxies: discrimination and calibration for medical diagnosis and treatment</article-title><source>J Am Med Inform Assoc</source><year>2025</year><month>01</month><day>1</day><volume>32</volume><issue>1</issue><fpage>139</fpage><lpage>149</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocae254</pub-id><pub-id pub-id-type="medline">39396184</pub-id></nlm-citation></ref><ref id="ref67"><label>67</label><nlm-citation citation-type="web"><article-title>AMIA artificial intelligence evaluation showcase</article-title><source>American Medical Informatics Association</source><access-date>2026-09-14</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://web.archive.org/web/20250908082941/https://amia.org/education-events/amia-artificial-intelligence-evaluation-showcase">https://web.archive.org/web/20250908082941/https://amia.org/education-events/amia-artificial-intelligence-evaluation-showcase</ext-link></comment></nlm-citation></ref><ref id="ref68"><label>68</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Minaee</surname><given-names>S</given-names> </name><name name-style="western"><surname>Mikolov</surname><given-names>T</given-names> </name><name name-style="western"><surname>Nikzad</surname><given-names>N</given-names> </name><name name-style="western"><surname>Chenaghlu</surname><given-names>M</given-names> </name><name name-style="western"><surname>Socher</surname><given-names>R</given-names> </name><name name-style="western"><surname>Amatriain</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Large language models: a survey</article-title><source>arXiv</source><comment>Preprint posted online on  Feb 9, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2402.06196</pub-id></nlm-citation></ref><ref id="ref69"><label>69</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shyr</surname><given-names>C</given-names> </name><name name-style="western"><surname>Ren</surname><given-names>B</given-names> </name><name name-style="western"><surname>Hsu</surname><given-names>CY</given-names> </name><etal/></person-group><article-title>A statistical framework for evaluating the repeatability and reproducibility of large language models</article-title><source>medRxiv</source><year>2025</year><month>11</month><day>4</day><fpage>2025.08.06.25333170</fpage><pub-id pub-id-type="doi">10.1101/2025.08.06.25333170</pub-id><pub-id pub-id-type="medline">41282788</pub-id></nlm-citation></ref><ref id="ref70"><label>70</label><nlm-citation citation-type="web"><article-title>Usability</article-title><source>Computer Security Resource Center, National Institute of Standards and Technology</source><access-date>2026-09-14</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://csrc.nist.gov/glossary/term/usability">https://csrc.nist.gov/glossary/term/usability</ext-link></comment></nlm-citation></ref><ref id="ref71"><label>71</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dehghani Mahmoodabadi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Dehghan</surname><given-names>H</given-names> </name><name name-style="western"><surname>Kimiafar</surname><given-names>K</given-names> </name><name name-style="western"><surname>Karami</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mousavi Baigi</surname><given-names>SF</given-names> </name><name name-style="western"><surname>Farsi Mehrabadi</surname><given-names>M</given-names> </name></person-group><article-title>Usability evaluation methods for hospital information systems: a systematic review</article-title><source>BMC Health Serv Res</source><year>2025</year><month>11</month><day>21</day><volume>25</volume><issue>1</issue><fpage>1540</fpage><pub-id pub-id-type="doi">10.1186/s12913-025-13745-y</pub-id><pub-id pub-id-type="medline">41272660</pub-id></nlm-citation></ref><ref id="ref72"><label>72</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Markatou</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kennedy</surname><given-names>O</given-names> </name><name name-style="western"><surname>Brachmann</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mukhopadhyay</surname><given-names>R</given-names> </name><name name-style="western"><surname>Dharia</surname><given-names>A</given-names> </name><name name-style="western"><surname>Talal</surname><given-names>AH</given-names> </name></person-group><article-title>Social determinants of health derived from people with opioid use disorder: Improving data collection, integration and use with cross-domain collaboration and reproducible, data-centric, notebook-style workflows</article-title><source>Front Med (Lausanne)</source><year>2023</year><volume>10</volume><fpage>1076794</fpage><pub-id pub-id-type="doi">10.3389/fmed.2023.1076794</pub-id><pub-id pub-id-type="medline">36936205</pub-id></nlm-citation></ref><ref id="ref73"><label>73</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Keyes</surname><given-names>T</given-names> </name><name name-style="western"><surname>Callahan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Pandya</surname><given-names>AS</given-names> </name><etal/></person-group><article-title>Why and how to monitor deployed AI systems in health care</article-title><source>NEJM Catalyst</source><year>2026</year><month>05</month><day>20</day><volume>7</volume><issue>6</issue><pub-id pub-id-type="doi">10.1056/CAT.25.0372</pub-id></nlm-citation></ref><ref id="ref74"><label>74</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hussein</surname><given-names>R</given-names> </name><name name-style="western"><surname>Zink</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ramadan</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Advancing healthcare AI governance through a comprehensive maturity model based on systematic review</article-title><source>NPJ Digit Med</source><year>2026</year><month>02</month><day>11</day><volume>9</volume><issue>1</issue><fpage>236</fpage><pub-id pub-id-type="doi">10.1038/s41746-026-02418-7</pub-id><pub-id pub-id-type="medline">41673321</pub-id></nlm-citation></ref><ref id="ref75"><label>75</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>El Arab</surname><given-names>RA</given-names> </name><name name-style="western"><surname>Mustafa</surname><given-names>MH</given-names> </name><name name-style="western"><surname>Almagharbeh</surname><given-names>WT</given-names> </name><etal/></person-group><article-title>Beyond model development in healthcare AI: post-development robustness, post-deployment monitoring, and lifecycle governance-a scoping review of reviews</article-title><source>Healthcare (Basel)</source><year>2026</year><month>05</month><day>25</day><volume>14</volume><issue>11</issue><fpage>1459</fpage><pub-id pub-id-type="doi">10.3390/healthcare14111459</pub-id><pub-id pub-id-type="medline">42278712</pub-id></nlm-citation></ref><ref id="ref76"><label>76</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>O&#x2019;Donnell</surname><given-names>B</given-names> </name><name name-style="western"><surname>Gupta</surname><given-names>V</given-names> </name></person-group><source>Continuous Quality Improvement</source><year>2026</year><access-date>2026-09-24</access-date><publisher-name>StatPearls</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/books/NBK559239/">https://www.ncbi.nlm.nih.gov/books/NBK559239/</ext-link></comment></nlm-citation></ref><ref id="ref77"><label>77</label><nlm-citation citation-type="web"><article-title>Ethics and governance of artificial intelligence for health: guidance on large multi-modal models</article-title><source>World Health Organization</source><year>2024</year><access-date>2026-09-14</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.who.int/publications/i/item/9789240084759">https://www.who.int/publications/i/item/9789240084759</ext-link></comment></nlm-citation></ref><ref id="ref78"><label>78</label><nlm-citation citation-type="web"><article-title>Regulation (EU) 2024/1689 of the european parliament and of the council of 13 june 2024 laying down harmonised rules on artificial intelligence and amending regulations (EC) no 300/2008, (EU) no 167/2013, (EU) no 168/2013, (EU) 2018/858, (EU) 2018/1139 and (EU) 2019/2144 and directives 2014/90/EU, (EU) 2016/797 and (EU) 2020/1828 (artificial intelligence act)</article-title><source>EUR-Lex</source><access-date>2026-09-14</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://eur-lex.europa.eu/eli/reg/2024/1689/oj">https://eur-lex.europa.eu/eli/reg/2024/1689/oj</ext-link></comment></nlm-citation></ref><ref id="ref79"><label>79</label><nlm-citation citation-type="web"><article-title>EU AI act: first regulation on artificial intelligence</article-title><source>European Parliament</source><access-date>2026-09-14</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.europarl.europa.eu/topics/en/article/20230601STO93804/eu-ai-act-first-regulation-on-artificial-intelligence">https://www.europarl.europa.eu/topics/en/article/20230601STO93804/eu-ai-act-first-regulation-on-artificial-intelligence</ext-link></comment></nlm-citation></ref><ref id="ref80"><label>80</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Madiega</surname><given-names>T</given-names> </name></person-group><article-title>Artificial intelligence act</article-title><year>2024</year><access-date>2026-09-14</access-date><publisher-name>European Parliamentary Research Service</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.europarl.europa.eu/RegData/etudes/BRIE/2021/698792/EPRS_BRI(2021)698792_EN.pdf">https://www.europarl.europa.eu/RegData/etudes/BRIE/2021/698792/EPRS_BRI(2021)698792_EN.pdf</ext-link></comment></nlm-citation></ref><ref id="ref81"><label>81</label><nlm-citation citation-type="web"><article-title>California senate bill 53 (2025-2026 regular session): artificial intelligence models: large developers</article-title><source>Legi Scan</source><year>2025</year><access-date>2026-09-14</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://legiscan.com/CA/bill/SB53/2025">https://legiscan.com/CA/bill/SB53/2025</ext-link></comment></nlm-citation></ref><ref id="ref82"><label>82</label><nlm-citation citation-type="web"><article-title>Center for drug evaluation and research artificial intelligence for drug development</article-title><source>US Food and Drug Administration</source><year>2025</year><access-date>2026-09-14</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.fda.gov/about-fda/center-drug-evaluation-and-research-cder/artificial-intelligence-drug-development">https://www.fda.gov/about-fda/center-drug-evaluation-and-research-cder/artificial-intelligence-drug-development</ext-link></comment></nlm-citation></ref><ref id="ref83"><label>83</label><nlm-citation citation-type="web"><article-title>Ensuring a national policy framework for artificial intelligence</article-title><source>Fed Regist</source><year>2025</year><access-date>2026-09-14</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.federalregister.gov/documents/2025/12/16/2025-23092/ensuring-a-national-policy-framework-for-artificial-intelligence">https://www.federalregister.gov/documents/2025/12/16/2025-23092/ensuring-a-national-policy-framework-for-artificial-intelligence</ext-link></comment></nlm-citation></ref><ref id="ref84"><label>84</label><nlm-citation citation-type="web"><article-title>Governor Hochul signs nation-leading legislation to require AI frameworks for AI frontier models</article-title><source>New York State Governor&#x2019;s Office</source><year>2025</year><access-date>2026-09-14</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.governor.ny.gov/news/governor-hochul-signs-nation-leading-legislation-require-ai-frameworks-ai-frontier-models">https://www.governor.ny.gov/news/governor-hochul-signs-nation-leading-legislation-require-ai-frameworks-ai-frontier-models</ext-link></comment></nlm-citation></ref><ref id="ref85"><label>85</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kolbinger</surname><given-names>FR</given-names> </name><name name-style="western"><surname>Veldhuizen</surname><given-names>GP</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Truhn</surname><given-names>D</given-names> </name><name name-style="western"><surname>Kather</surname><given-names>JN</given-names> </name></person-group><article-title>Reporting guidelines in medical artificial intelligence: a systematic review and meta-analysis</article-title><source>Commun Med (Lond)</source><year>2024</year><month>04</month><day>11</day><volume>4</volume><issue>1</issue><fpage>71</fpage><pub-id pub-id-type="doi">10.1038/s43856-024-00492-0</pub-id><pub-id pub-id-type="medline">38605106</pub-id></nlm-citation></ref><ref id="ref86"><label>86</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Li</surname><given-names>N</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Guidelines, consensus statements, and standards for the use of artificial intelligence in medicine: systematic review</article-title><source>J Med Internet Res</source><year>2023</year><month>11</month><day>22</day><volume>25</volume><fpage>e46089</fpage><pub-id pub-id-type="doi">10.2196/46089</pub-id><pub-id pub-id-type="medline">37991819</pub-id></nlm-citation></ref><ref id="ref87"><label>87</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sim</surname><given-names>I</given-names> </name><name name-style="western"><surname>Cassel</surname><given-names>C</given-names> </name></person-group><article-title>The ethics of relational AI &#x2014; expanding and implementing the Belmont principles</article-title><source>N Engl J Med</source><year>2024</year><month>07</month><day>18</day><volume>391</volume><issue>3</issue><fpage>193</fpage><lpage>196</lpage><pub-id pub-id-type="doi">10.1056/NEJMp2314771</pub-id><pub-id pub-id-type="medline">39007542</pub-id></nlm-citation></ref><ref id="ref88"><label>88</label><nlm-citation citation-type="web"><article-title>Augmented intelligence development, deployment, and use in health care</article-title><source>American Medical Association</source><access-date>2026-09-14</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.ama-assn.org/system/files/ama-ai-principles.pdf">https://www.ama-assn.org/system/files/ama-ai-principles.pdf</ext-link></comment></nlm-citation></ref><ref id="ref89"><label>89</label><nlm-citation citation-type="web"><article-title>Blueprint for trustworthy AI implementation guidance and assurance for healthcare</article-title><source>Coalition for Health AI</source><year>2023</year><access-date>2026-09-14</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://assets.ctfassets.net/7s4afyr9pmov/4AXIWGIlcrjWDaW2ueTaRS/f98e5cb2528187635895cce6ba5ec309/Blueprint_for_Trustworthy_AI.pdf">https://assets.ctfassets.net/7s4afyr9pmov/4AXIWGIlcrjWDaW2ueTaRS/f98e5cb2528187635895cce6ba5ec309/Blueprint_for_Trustworthy_AI.pdf</ext-link></comment></nlm-citation></ref><ref id="ref90"><label>90</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Uszkoreit</surname><given-names>H</given-names> </name><name name-style="western"><surname>Du</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Fan</surname><given-names>W</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>D</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>J</given-names> </name></person-group><article-title>Explainable AI: a brief survey on history, research areas, approaches and challenges</article-title><conf-name>Natural Language Processing and Chinese Computing: 8th CCF International Conference, NLPCC 2019</conf-name><conf-date>Oct 9-14, 2019</conf-date><conf-loc>Dunhuang, China</conf-loc><fpage>563</fpage><lpage>574</lpage><pub-id pub-id-type="doi">10.1007/978-3-030-32236-6_51</pub-id></nlm-citation></ref><ref id="ref91"><label>91</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Young</surname><given-names>CC</given-names> </name><name name-style="western"><surname>Enichen</surname><given-names>E</given-names> </name><name name-style="western"><surname>Rao</surname><given-names>A</given-names> </name><name name-style="western"><surname>Succi</surname><given-names>MD</given-names> </name></person-group><article-title>Racial, ethnic, and sex bias in large language model opioid recommendations for pain management</article-title><source>Pain</source><year>2025</year><month>03</month><day>1</day><volume>166</volume><issue>3</issue><fpage>511</fpage><lpage>517</lpage><pub-id pub-id-type="doi">10.1097/j.pain.0000000000003388</pub-id><pub-id pub-id-type="medline">39283333</pub-id></nlm-citation></ref><ref id="ref92"><label>92</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Glasgow</surname><given-names>RE</given-names> </name><name name-style="western"><surname>Harden</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Gaglio</surname><given-names>B</given-names> </name><etal/></person-group><article-title>RE-AIM planning and evaluation framework: adapting to new science and practice with a 20-year review</article-title><source>Front Public Health</source><year>2019</year><volume>7</volume><fpage>64</fpage><pub-id pub-id-type="doi">10.3389/fpubh.2019.00064</pub-id><pub-id pub-id-type="medline">30984733</pub-id></nlm-citation></ref><ref id="ref93"><label>93</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Damschroder</surname><given-names>LJ</given-names> </name><name name-style="western"><surname>Reardon</surname><given-names>CM</given-names> </name><name name-style="western"><surname>Widerquist</surname><given-names>MAO</given-names> </name><name name-style="western"><surname>Lowery</surname><given-names>J</given-names> </name></person-group><article-title>The updated consolidated framework for implementation research based on user feedback</article-title><source>Implementation Sci</source><year>2022</year><volume>17</volume><issue>1</issue><fpage>75</fpage><pub-id pub-id-type="doi">10.1186/s13012-022-01245-0</pub-id></nlm-citation></ref><ref id="ref94"><label>94</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Olawade</surname><given-names>DB</given-names> </name><name name-style="western"><surname>Plabon</surname><given-names>SB</given-names> </name><name name-style="western"><surname>Ojo</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ogunbona</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Makanjuola</surname><given-names>BD</given-names> </name><name name-style="western"><surname>Olasilola</surname><given-names>OR</given-names> </name></person-group><article-title>Human in the loop artificial intelligence in healthcare: applications, outcomes, and implementation challenges</article-title><source>Int J Med Inform</source><year>2026</year><month>06</month><day>15</day><volume>213</volume><fpage>106362</fpage><pub-id pub-id-type="doi">10.1016/j.ijmedinf.2026.106362</pub-id><pub-id pub-id-type="medline">41740273</pub-id></nlm-citation></ref><ref id="ref95"><label>95</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lazaros</surname><given-names>K</given-names> </name><name name-style="western"><surname>Vrahatis</surname><given-names>AG</given-names> </name><name name-style="western"><surname>Kotsiantis</surname><given-names>S</given-names> </name></person-group><article-title>Human-in-the-loop artificial intelligence: a systematic review of concepts, methods, and applications</article-title><source>Entropy (Basel)</source><year>2026</year><month>03</month><day>26</day><volume>28</volume><issue>4</issue><fpage>377</fpage><pub-id pub-id-type="doi">10.3390/e28040377</pub-id><pub-id pub-id-type="medline">42072503</pub-id></nlm-citation></ref><ref id="ref96"><label>96</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>L&#x00F3;pez-&#x00DA;beda</surname><given-names>P</given-names> </name><name name-style="western"><surname>Mart&#x00ED;n-Noguerol</surname><given-names>T</given-names> </name><name name-style="western"><surname>Luna</surname><given-names>A</given-names> </name></person-group><article-title>Environmental and economic costs behind LLMs</article-title><source>Int J CARS</source><year>2026</year><volume>21</volume><issue>3</issue><fpage>661</fpage><lpage>663</lpage><pub-id pub-id-type="doi">10.1007/s11548-026-03568-5</pub-id></nlm-citation></ref><ref id="ref97"><label>97</label><nlm-citation citation-type="web"><article-title>Developing methods for clustering longitudinal mixed-type data for comparative effectiveness research</article-title><source>University at Buffalo</source><access-date>2026-09-14</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://ubwp.buffalo.edu/clustllm4cer/">https://ubwp.buffalo.edu/clustllm4cer/</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>A brief history of natural language processing and evolution timeline of the field of natural language processing.</p><media xlink:href="jmir_v28i1e97131_app1.docx" xlink:title="DOCX File, 619 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Details on large language models and brief literature.</p><media xlink:href="jmir_v28i1e97131_app2.docx" xlink:title="DOCX File, 29 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Evaluation of CREATE framework.</p><media xlink:href="jmir_v28i1e97131_app3.docx" xlink:title="DOCX File, 24 KB"/></supplementary-material></app-group></back></article>