<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="review-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e98551</article-id><article-id pub-id-type="doi">10.2196/98551</article-id><article-categories><subj-group subj-group-type="heading"><subject>Review</subject></subj-group></article-categories><title-group><article-title>Generative Artificial Intelligence for Qualitative Methods in Health Research: Rapid Review</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Gierbolini-Rivera</surname><given-names>Ra&#x00FA;l D</given-names></name><degrees>MPH</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Franco Silva</surname><given-names>Milena</given-names></name><degrees>MUP</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Augusto de Paula da Silva</surname><given-names>Alexandre</given-names></name><degrees>MS, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Brownson</surname><given-names>Ross C</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Parra</surname><given-names>Diana C</given-names></name><degrees>MPH, PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kepper</surname><given-names>Maura M</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Eyler</surname><given-names>Amy A</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib></contrib-group><aff id="aff1"><institution>Prevention Research Center, Bursky School of Public Health, Washington University in St. Louis</institution><addr-line>One Brookings Drive</addr-line><addr-line>St. Louis</addr-line><addr-line>MO</addr-line><country>United States</country></aff><aff id="aff2"><institution>Alvin J. Siteman Cancer Center, School of Medicine, Washington University in St. Louis</institution><addr-line>St. Louis</addr-line><addr-line>MO</addr-line><country>United States</country></aff><aff id="aff3"><institution>Brown School, Washington University in St. Louis</institution><addr-line>St. Louis</addr-line><addr-line>MO</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Verran</surname><given-names>Deborah</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Pratomo</surname><given-names>Hadi</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Howe</surname><given-names>Nicola</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Ra&#x00FA;l D Gierbolini-Rivera, MPH, Prevention Research Center, Bursky School of Public Health, Washington University in St. Louis, One Brookings Drive, St. Louis, MO, 63130, United States, 1 7874294475; <email>g.raul@wustl.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>31</day><month>8</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e98551</elocation-id><history><date date-type="received"><day>16</day><month>04</month><year>2026</year></date><date date-type="rev-recd"><day>29</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>10</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Ra&#x00FA;l D Gierbolini-Rivera, Milena Franco Silva, Alexandre Augusto de Paula da Silva, Ross C Brownson, Diana C Parra, Maura M Kepper, Amy A Eyler. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 31.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e98551"/><abstract><sec><title>Background</title><p>Generative AI (GAI) is rapidly transforming research practices, including qualitative methods in health research. While these tools offer efficiency in processing large volumes of textual data, concerns remain regarding their methodological rigor, interpretive capacity, equity, and ethical implications.</p></sec><sec><title>Objective</title><p>This rapid review aimed to synthesize the current evidence on the use of GAI in health-related qualitative research, focusing on its applications, performance relative to human analysis, and implications for rigor, ethics, and equity.</p></sec><sec sec-type="methods"><title>Methods</title><p>We conducted a rapid review following Joanna Briggs Institute and PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) guidelines. Peer-reviewed studies published between 2022 and December 2025 were identified through searches in PubMed, Web of Science, and Scopus. Eligible studies included qualitative or mixed methods research that used GAI tools (eg, ChatGPT, Gemini, and Claude) during qualitative analysis, including studies that compared GAI-generated outputs with human researchers, coders, or traditional qualitative analytic approaches. Data were extracted using a structured template and synthesized descriptively. Study quality was assessed using the Critical Appraisal Skills Programme (CASP) checklist. This rapid review was registered with the International Prospective Register of Systematic Reviews (PROSPERO; CRD420261280832). The review adhered to the registered PROSPERO protocol; no deviations occurred.</p></sec><sec sec-type="results"><title>Results</title><p>A total of 42 studies met the inclusion criteria; 71.4% (n=30) were published in 2025, and 81% (n=34) used qualitative designs. Thematic analysis (n=20, 40%) and content analysis (n=10, 20%) were the most common qualitative approaches. GAI was most applied during data familiarization, coding, and theme development, with ChatGPT being the most frequently reported GAI, accounting for nearly two-thirds of all model occurrences (n=37, 62.7%). Among studies evaluating GAI performance relative to human qualitative analysis, performance was strongest in inductive thematic and content analyses, with agreement often exceeding 80% for descriptive themes but dropping to approximately 30% for culturally nuanced themes. Several studies reported time to complete analyses up to 97% faster than human-led analyses. However, performance declined for reflexive and theory-driven analyses, particularly when interpreting culturally nuanced or emotionally complex data. Across studies, GAI improved efficiency but frequently produced superficial interpretations, misapplied theoretical frameworks, and generated occasional inaccuracies, including fabricated quotes. Human oversight was consistently identified as essential to ensure validity, contextual accuracy, and ethical integrity. Concerns related to bias, transparency, and data privacy were widely reported.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>GAI can effectively support early-stage qualitative analysis and enhance efficiency in health research; however, it cannot replace the interpretive and reflexive functions central to qualitative inquiry. A hybrid human-AI approach is recommended, in which GAI assists with data processing while researchers retain responsibility for interpretation, contextualization, and ethical oversight. Future research should prioritize developing guidelines that address equity, transparency, and responsible integration of GAI into qualitative methodologies.</p></sec></abstract><kwd-group><kwd>generative AI</kwd><kwd>qualitative methods</kwd><kwd>large language models</kwd><kwd>health</kwd><kwd>evidence synthesis</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Qualitative research is foundational in public health, health policy, and the social sciences because it explores experiences, meanings, processes, and contexts that shape health using nonnumerical data [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. Common data collection methods include interviews, focus groups, participant observation, and textual analysis [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. With flexible and iterative designs, these methods enable researchers to understand how people make health-related decisions and interpret the symbolic meanings embedded in their words and actions [<xref ref-type="bibr" rid="ref1">1</xref>]. This reflexive process can also be applied to existing documents, such as policies, media texts, social media posts, and legislation, to uncover contextual influences and intentions. Major methodological approaches include thematic analysis (identifying themes), grounded theory (generating theory from patterns), framework analysis (applying structured models), content analysis (categorizing text), and ethnography (describing cultures and social worlds) [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. Qualitative research identifies barriers and facilitators to health, informs intervention development, strengthens mixed methods designs, and deepens understanding of the social and contextual forces shaping health behaviors and outcomes [<xref ref-type="bibr" rid="ref1">1</xref>].</p><p>The rapid emergence of generative AI (GAI) has accelerated its use in qualitative research [<xref ref-type="bibr" rid="ref3">3</xref>]. GAI refers to a subset of AI that creates new content by learning patterns from existing data [<xref ref-type="bibr" rid="ref1">1</xref>]. Tools such as ChatGPT (OpenAI), Claude (Anthropic), and Gemini (Google) are transformer-based large language models (LLMs), a type of GAI trained on large volumes of text data that can generate human-like, semantically coherent responses and process large qualitative datasets [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref4">4</xref>]. Unlike traditional machine learning models focused on prediction or classification, GAI produces novel outputs, including text, code, or images, based on statistical pattern learning [<xref ref-type="bibr" rid="ref5">5</xref>]. These tools can identify themes, sentiments, and trends; generate coding schemes; transcribe interviews; analyze audio or video data; and even support structured approaches such as grounded theory [<xref ref-type="bibr" rid="ref1">1</xref>]. However, the literature highlights important limitations. GAI systems also struggle with subcultural slang and culturally specific discourse, posing concerns for global and equity-focused public health research [<xref ref-type="bibr" rid="ref1">1</xref>]. Jowsey et al [<xref ref-type="bibr" rid="ref6">6</xref>] argue that GAI lacks true reflexive capacity because it relies on statistical prediction rather than genuine comprehension of meaning, context, and human experience. Transformer-based models frequently hallucinate, generate nonexistent information, reproduce training data biases, and pose privacy risks when used in cloud-based systems [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref4">4</xref>].</p><p>GAI tools are integrated into multiple stages of qualitative research. AI-enabled computer-assisted qualitative data analysis software (CAQDAS) platforms (NVivo, ATLAS.ti, and MAXQDA) and general-purpose LLMs (ChatGPT, Claude, Gemini, and Bard) are used to automate transcription, generate word and concept clouds, auto-code transcripts, produce summary themes, cross-reference codes, and create tables and figures [<xref ref-type="bibr" rid="ref1">1</xref>]. Researchers have also applied prompt engineering strategies, such as explicit task instructions, contextual background, output templates, role-based prompting, and transparency directives, to improve LLM performance in thematic analysis [<xref ref-type="bibr" rid="ref4">4</xref>]. Zhang et al [<xref ref-type="bibr" rid="ref4">4</xref>] found that guided, transparent prompting increased researchers&#x2019; trust and shifted attitudes from skepticism to cautious acceptance through iterative refinement. Performance outcomes across different GAI tools, however, are mixed. For example, one study by Prescott et al [<xref ref-type="bibr" rid="ref7">7</xref>] found that themes generated by GAI (ChatGPT and Bard, now called &#x201C;Gemini&#x201D;) were consistent with 71% of themes produced by human analysts following inductive thematic analysis. However, consistency dropped to 50% to 58% for deductive thematic analysis, with intercoder reliability ranging from fair to moderate. Sakaguchi et al [<xref ref-type="bibr" rid="ref8">8</xref>] reported over 80% agreement on descriptive themes using ChatGPT-4, but only approximately 30% for culturally nuanced themes. Overall, GAI tools help accelerate coding and early theme development but still lack the contextual sensitivity and interpretive depth required for high-quality qualitative analysis [<xref ref-type="bibr" rid="ref1">1</xref>].</p><p>Concerns about methodological rigor and transparency are growing. Monforte [<xref ref-type="bibr" rid="ref9">9</xref>] warns that GAI tools may reduce qualitative inquiry to pattern recognition, prioritizing speed over reflection. LLMs are seen as &#x201C;black boxes,&#x201D; conflicting with qualitative research&#x2019;s transparency, reflexivity, and reliability [<xref ref-type="bibr" rid="ref9">9</xref>]. Studies warn that reliance on GAI could undermine deep thinking, critical reading, and reflexivity, raising questions about whether AI can genuinely interpret data or reinforce biases [<xref ref-type="bibr" rid="ref10">10</xref>-<xref ref-type="bibr" rid="ref12">12</xref>]. Sigman and Bilinkis [<xref ref-type="bibr" rid="ref13">13</xref>] call this &#x201C;cognitive sedentarism,&#x201D; where critical thinking declines and GAI outputs are accepted uncritically. Ethical and privacy issues are significant; cloud-based AI tools may store data externally or use it for training, risking confidentiality and participant rights [<xref ref-type="bibr" rid="ref14">14</xref>]. Broader social issues, such as exploitative labor and ecological impacts, complicate GAI use [<xref ref-type="bibr" rid="ref14">14</xref>]. Universities are creating governance frameworks, such as the University of Texas at Austin&#x2019;s pilot of Grammarly&#x2019;s AI with data protections and its &#x201C;Faculty Guide,&#x201D; and the University of Central Florida&#x2019;s AI conference, which brings together higher education professionals to share best practices, ethical considerations, and pedagogical strategies for integrating AI into academia [<xref ref-type="bibr" rid="ref15">15</xref>]. Vanderbilt University&#x2019;s &#x201C;walled-garden&#x201D; AI systems keep inputs within institutional infrastructure. These measures aim to balance innovation with ethics, privacy, and integrity [<xref ref-type="bibr" rid="ref15">15</xref>].</p><p>Since ChatGPT&#x2019;s public release in late 2022, the literature has emphasized the need for human insight, interpretation, and validation in qualitative research using GAI. Many recent studies have directly evaluated GAI-generated outputs against human qualitative researchers or traditional qualitative analytic approaches, creating a growing evidence base regarding the strengths and limitations of GAI-assisted qualitative analysis. However, some scholars argue that GAI conflicts with reflexive methodologies that require meaning-making and interpretive rigor and should be avoided. With evolving GAI technologies and debates about their use, a synthesis of evidence is needed. No systematic review exists on the use of GAI in health-related qualitative methods. As GAI is already affecting health researchers, a rapid review is necessary to provide insights into its applications. This review aimed to (1) describe GAI integration; (2) compare performance with humans or computer-assisted qualitative data analysis softComputer-Assisted Qualitative Data Analysis Software; (3) assess rigor, reproducibility, ethics, and equity; and (4) identify gaps, risks, and best practices. Terms such as LLMs and GAI tools are used interchangeably to refer to related concepts. The findings of this study can broaden understanding of the current use of AI in health research, as well as the benefits and trade-offs of using it in qualitative methods, especially amid the increasing integration of AI across various fields, including academia and research.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Overview</title><p>We conducted a rapid review of journal articles on GAI in health-related qualitative methods, following Joanna Briggs Institute (JBI) and PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) guidelines (<xref ref-type="supplementary-material" rid="app5">Checklist 1</xref>) [<xref ref-type="bibr" rid="ref16">16</xref>]. This review was registered in the International Prospective Register of Systematic Reviews (PROSPERO; ID: CRD420261280832; National Institute for Health and Care Research, 2026). A rapid review is a form of knowledge synthesis that accelerates the evidence gathering process to inform urgent decision-making, differentiating itself from a systematic review by streamlining specific methodological steps, such as searching fewer databases or limiting gray literature, to produce results in a faster manner [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. Conducting a rapid review was necessary because GAI is being rapidly integrated into qualitative research and may have implications for research rigor.</p></sec><sec id="s2-2"><title>Inclusion Criteria</title><p>Studies were included if they met the SPIDER (sample, phenomenon of interest, design, evaluation, research type) framework, which is well-suited for qualitative and mixed-methods rapid reviews [<xref ref-type="bibr" rid="ref18">18</xref>]. The criteria were as follows. First, the eligibility criteria included qualitative or mixed methods studies conducted in health-related settings (public health, health policy, community health, or related social science contexts with clear health relevance). Studies focused purely on clinical contexts (eg, medical transcription using AI) or education sector contexts were excluded. The review included studies from any health domain, such as chronic disease, health behaviors, and noncommunicable diseases, but excluded articles focused solely on clinical applications, such as medical documentation, surgical decisions, and clinical workflows. This exclusion was intentional because a substantial and rapidly growing body of literature already examines the use of GAI to support clinical workflows and health care operations. The aim of this rapid review was to focus specifically on the use of GAI within qualitative health research methodologies and analytic processes, rather than on clinical or operational applications of AI in health care. Second, the phenomenon of interest was empirical use of GAI tools (eg, ChatGPT, Claude, Gemini, and Llama-based models) in at least one qualitative research step. Conceptual papers and reviews were excluded. Third, the study design included qualitative studies, methodological demonstrations, case studies, and mixed methods or other empirical study designs, provided they contained identifiable qualitative components relevant to the review objectives. Fourth, the evaluation included at least one outcome related to the methodological or practical performance of GAI, including comparison with human researchers, coders, traditional qualitative analytic approaches, or outcomes related to transparency, ethics, or equity associated with its use. Fifth, the research type was limited to empirical, peer-reviewed articles. We included studies published since 2022, aligning with the mainstream adoption of GAI, and limited the search to English-language publications. We included studies from 2022 onward to capture the most recent developments in GAI and better understand emerging ethical and equity concerns.</p></sec><sec id="s2-3"><title>Search Strategy</title><p>We conducted the search across 3 databases (PubMed, Web of Science, and Scopus), finalizing it on December 11, 2025. These databases were chosen because they cover research in health, public health, and the social sciences, aligning well with the aims of this study. A Boolean logic syntax using all keyword combinations was used (see Appendix I in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The full search strategy is provided in Appendix II in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></sec><sec id="s2-4"><title>Study Screening</title><p>Initially, we screened studies by title and abstract, then conducted full-text screening of those that met the criteria. One reviewer screened all records, while a second reviewer independently sampled 30% of the records. Discrepancies were resolved by consensus among the 2 reviewers. All screening and agreement checks were documented in Rayyan, an online platform for systematic reviews [<xref ref-type="bibr" rid="ref19">19</xref>].</p></sec><sec id="s2-5"><title>Data Extraction and Synthesis</title><p>Data extraction was carried out in accordance with the JBI data extraction standards [<xref ref-type="bibr" rid="ref17">17</xref>]. A structured Excel data extraction template was adapted from the JBI data collection tools. To ensure consistency, the Excel template was pilot-tested by 2 reviewers on a small subset of studies, enabling the team to make refinements. The extracted data included bibliographic details (such as title, author, year, journal, citation, and study aim), health domain, study setting, study design, country, institutional affiliations, qualitative approach, GAI used, specific use of the AI tool, reliability of human versus GAI, evaluation metrics, main findings, and ethics related to the use of GAI in qualitative health research. All data were systematically extracted by one reviewer, with a second reviewer checking a randomized 30% subset. The results were reported in accordance with PRISMA guidelines; the synthesis did not require a minimum number of studies to be included or analyzed. Refer to Appendix III in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref> for the variables used in data extraction. To enhance readability and alignment, we present a series of synthesis tables in the <italic>Results</italic> section that summarize patterns across GAI methods, performance, ethics, and trade-offs in qualitative health research. The full study-level extraction table is provided in Appendix IV in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>.</p></sec><sec id="s2-6"><title>Assessment of Quality</title><p>All studies were rated for quality using the Critical Appraisal Skills Programme (CASP) checklist [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>]. For each item, the following categories were used: &#x201C;Yes,&#x201D; &#x201C;Somewhat,&#x201D; &#x201C;Can&#x2019;t Tell,&#x201D; or &#x201C;No.&#x201D; For the last item in the checklist, a narrative judgment of the value of the research was developed for each study. Regardless of the category in which each item was placed, all studies were included in the review, as recommended by the CASP guidelines [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>].</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Characteristics of the Included Studies and Main Findings</title><p>The systematic search identified 619 records (Scopus: n=212, 34.2%; PubMed: n=372, 60.1%; and Web of Science: n=35, 5.7%), as shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>. After removing 54 (8.7%) duplicates, 565 (91.3%) unique records underwent title and abstract screening, during which 487 (86.2%) were excluded. Seventy-eight (13.8%) papers proceeded to full-text review, and 42 (7.4%) met all inclusion criteria for this rapid review. The reasons for exclusion were the wrong setting (n=20, 55.6%), not related to health (n=8, 22.2%), the wrong publication type (n=4, 11%), not qualitative (n=2, 5.6%), and examining only AI perception (n=2, 5.6%).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) flowchart of the included studies (n=42).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e98551_fig01.png"/></fig><p><xref ref-type="table" rid="table1">Table 1</xref> presents that most studies were published in 2025 (n=30, 71.4%). The majority were qualitative studies (n=34, 81%), while the remainder were mixed methods or other study designs that contained identifiable qualitative components. Thematic analysis (20/50, 40%) and content analysis (10/50, 20%) were the most common methods, and in some studies, multiple qualitative approaches were used. The most referenced health domains across the studies were health communication, cancer, COVID-19, medical education, and mental health. Other, less common health domains included substance use, patient safety, climate change, LGBTQ+ health, and multiple clinical specialties.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Summary characteristics of the included studies.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Variables and categories</td><td align="left" valign="bottom">Values, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top">Year</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>2025</td><td align="left" valign="top">30 (71.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>2024</td><td align="left" valign="top">12 (28.6)</td></tr><tr><td align="left" valign="top">Study design</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Qualitative only</td><td align="left" valign="top">34 (81.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mixed methods</td><td align="left" valign="top">6 (14.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Cross-sectional with qualitative elements</td><td align="left" valign="top">1 (2.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Experimental and comparative study with qualitative elements</td><td align="left" valign="top">1 (2.4)</td></tr><tr><td align="left" valign="top">Qualitative approach<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Thematic analysis</td><td align="left" valign="top">20 (40.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Content analysis</td><td align="left" valign="top">10 (20.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Grounded theory</td><td align="left" valign="top">4 (8.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Constant comparative method</td><td align="left" valign="top">2 (4.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Framework analysis</td><td align="left" valign="top">2 (4.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No formal approach&#x2014;qualitative elements</td><td align="left" valign="top">2 (4.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Case study</td><td align="left" valign="top">1 (2.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Codebook-based coding</td><td align="left" valign="top">1 (2.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Autoethnography</td><td align="left" valign="top">1 (2.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Immersion or crystallization</td><td align="left" valign="top">1 (2.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Thematic narrative analysis</td><td align="left" valign="top">1 (2.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Qualitative comparative analysis</td><td align="left" valign="top">1 (2.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Qualitative description</td><td align="left" valign="top">1 (2.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Query-based analysis</td><td align="left" valign="top">1 (2.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>System dynamics or causal loop</td><td align="left" valign="top">1 (2.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Iterative thematic inquiry</td><td align="left" valign="top">1 (2.0)</td></tr><tr><td align="left" valign="top">GAI<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> model used<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Open AI (ChatGPT)</td><td align="left" valign="top">37 (62.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Meta Llama</td><td align="left" valign="top">6 (10.2)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Google Gemini</td><td align="left" valign="top">4 (6.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Google Gemma</td><td align="left" valign="top">2 (3.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Claude</td><td align="left" valign="top">2 (3.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Google Flan</td><td align="left" valign="top">2 (3.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Microsoft Copilot</td><td align="left" valign="top">2 (3.4)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Perplexity</td><td align="left" valign="top">1 (1.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Grok</td><td align="left" valign="top">1 (1.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>DeepSeek</td><td align="left" valign="top">1 (1.7)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral AI</td><td align="left" valign="top">1 (1.7)</td></tr><tr><td align="left" valign="top">Health domains<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Health communication</td><td align="left" valign="top">4 (9.5)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Cancer</td><td align="left" valign="top">3 (7.1)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>COVID-19</td><td align="left" valign="top">2 (4.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Medical education</td><td align="left" valign="top">2 (4.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mental health</td><td align="left" valign="top">2 (4.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Substance use</td><td align="left" valign="top">2 (4.8)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Other<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">27 (64.3)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>There was a total of 42 included studies. The variables of &#x201C;qualitative approach,&#x201D; &#x201C;GAI model used,&#x201D; and &#x201C;health domains&#x201D; may exceed 100% because there were multiple approaches in some studies.</p></fn><fn id="table1fn2"><p><sup>b</sup>GAI: generative AI.</p></fn><fn id="table1fn3"><p><sup>c</sup>In this other category, there were 27 distinct health domains: addiction, asthma, blindness, cardiovascular disease, climate change, clinical practice, emergency medicine, health care, HIV, hospice care, LGBTQ+ health, maternal health, medication management, nursing, nutrition, obesity, ophthalmology, patient safety, pharmacovigilance, public health, sacred moments, school psychology, sleep medicine, social media, surgery, telemental health care, urology, vaccine hesitancy, and well-being.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Narrative Synthesis</title><p>Across the included studies, GAI was used at multiple stages of qualitative analysis, with varying levels of effectiveness depending on the analytic task, methodological approach, and degree of human oversight. <xref ref-type="table" rid="table2">Table 2</xref> summarizes how GAI tools were integrated across the qualitative research stages, their typical uses, the human role, and key observations.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Integration of generative AI (GAI) across qualitative research stages<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Qualitative stage</td><td align="left" valign="bottom">Typical use of GAI</td><td align="left" valign="bottom">Human role</td><td align="left" valign="bottom">Key observations</td><td align="left" valign="bottom">Reference</td></tr></thead><tbody><tr><td align="left" valign="top">Familiarization</td><td align="left" valign="top">Summarization, surface pattern detection</td><td align="left" valign="top">Contextual verification</td><td align="left" valign="top">Across 6 studies evaluating primarily ChatGPT, Copilot, Mistral, and related LLMs<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup>, GAI-generated summaries were broadly similar to manual analyses. Four of the 6 studies reported that outputs were primarily descriptive and missed irony, emotional tone, or contextual nuance, while all 6 studies emphasized the need for human verification. Performance appeared strongest for summarization and surface-level familiarization tasks but varied across models when deeper contextual understanding was required.</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref22">22</xref>-<xref ref-type="bibr" rid="ref27">27</xref>]</td></tr><tr><td align="left" valign="top">Inductive coding</td><td align="left" valign="top">First-pass code generation</td><td align="left" valign="top">Code refinement</td><td align="left" valign="top">Across 2 studies evaluating ChatGPT-3.5 and ChatGPT-4, GAI-generated themes were consistent with those identified by human analysts, with agreement exceeding 80% in one study. Both studies concluded that human review remained necessary. While ChatGPT demonstrated strong first-pass coding capability, evidence was insufficient to determine whether comparable performance would be achieved across other GAI models.</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref28">28</xref>]</td></tr><tr><td align="left" valign="top">Deductive coding</td><td align="left" valign="top">Codebook application</td><td align="left" valign="top">Error correction</td><td align="left" valign="top">One study (1/2) evaluating Llama 3 70B reported high accuracy for binary and concrete codes but weaker performance on behavioral and interpersonal constructs, including frequent false positives. Both studies recommended hybrid human-AI workflows to improve accuracy, contextual interpretation, and bias mitigation. Findings suggest stronger performance for structured deductive coding than complex interpretive coding tasks.</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]</td></tr><tr><td align="left" valign="top">Theme development</td><td align="left" valign="top">Code clustering</td><td align="left" valign="top">Theoretical interpretation</td><td align="left" valign="top">Across 4 studies evaluating various variants of ChatGPT, and Mistral, GAI-generated themes were broadly similar to those identified through manual analysis. Three of 4 studies reported themes more descriptive and less interpretive than those produced by humans, 3 of 4 studies reported forced or misapplied theoretical interpretations. These limitations were observed across multiple models rather than being unique to a single model, underscoring the continued need for human oversight during theme development.</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref32">32</xref>]</td></tr><tr><td align="left" valign="top">Quote extraction</td><td align="left" valign="top">Illustrative excerpts</td><td align="left" valign="top">Hallucination checks</td><td align="left" valign="top">Across the 5 studies evaluating primarily ChatGPT variants and a local Mistral-7B model, GAI-generated outputs included usable illustrative quotes. However, 4 of 5 studies reported fabricated, paraphrased, or inaccurately attributed quotes requiring manual correction. All 5 studies emphasized expert oversight and recommended hybrid human-AI workflows for quote verification and reporting.</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref33">33</xref>-<xref ref-type="bibr" rid="ref35">35</xref>]</td></tr><tr><td align="left" valign="top">Interpretation</td><td align="left" valign="top">Theory linking and implications</td><td align="left" valign="top">Reflexive sense making</td><td align="left" valign="top">Across 4 studies evaluating ChatGPT variants and a local Mistral-7B model, GAI performance declined when analyses required cultural, emotional, or theoretical interpretation. All 4 studies reported weaker theoretical grounding, reduced contextual insight, or difficulty with nuanced interpretation compared with human researchers. Although the severity of these limitations varied by model, no evaluated model consistently matched human performance in theory-driven interpretive analysis.</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref36">36</xref>]</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>ChatGPT variants were the most frequently evaluated systems across studies included in this synthesis. Evidence for Copilot, Bard-Gemini, Claude, Mistral, Llama, DeepSeek, and other models was more limited. Consequently, findings summarized at the level of &#x201C;GAI&#x201D; should not be interpreted as evidence that all systems perform similarly. where available, model-specific findings and differences are noted.</p></fn><fn id="table2fn2"><p><sup>b</sup>LLM: large language model.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Integration of GAI Across Qualitative Research Stages</title><p>During data familiarization, GAI tools were commonly used to summarize large volumes of text and identify surface-level patterns. Across studies, AI-generated summaries aligned broadly with manual familiarization; however, they consistently struggled with irony, emotional tone, and contextual nuance, particularly in culturally embedded data. As presented in <xref ref-type="table" rid="table2">Table 2</xref>, researchers emphasized the importance of human contextual verification at this early stage to prevent misinterpretation. For inductive coding, GAI tools were frequently applied as first-pass coders. Most studies reported strong concordance between AI-generated and human-generated codes, particularly for manifest content and frequently occurring concepts. Inductive codes produced by GAI tools were generally considered reliable and useful for accelerating early analytic phases; however, all studies underscored the need for continued human refinement and validation. In contrast, deductive coding revealed clearer boundaries to GAI performance.</p><p>While GAI tools achieved high accuracy for binary or concrete codes, performance declined for behavioral, interpersonal, and theoretically complex constructs. False positives and overapplication of codebook categories were common, reinforcing the value of hybrid workflows in which GAI tools accelerate code application, but humans correct errors to ensure conceptual fidelity. During theme development and interpretation, GAI tools were effective at clustering codes and identifying broad thematic structures but showed inconsistent ability to differentiate valence, capture latent meanings, or apply theory appropriately. Several studies reported that AI-generated themes resembled those produced by experts at a surface level; however, they lacked reflexive depth or introduced forced theoretical connections. These findings highlight that while GAI tools can support the initial stages of analysis, theoretical interpretation remains fundamentally human driven. GAI tools were often used for quote extraction; however, hallucinated or paraphrased quotes persisted even in studies using verification workflows.</p></sec><sec id="s3-4"><title>Reliability and Evaluation of GAI Relative to Human Analysis</title><p>Across the included studies, reliability and performance were commonly assessed by comparing GAI-generated codes, categories, themes, or summaries with those produced by human researchers. Measures of agreement included percent agreement, Cohen &#x03BA;, Krippendorff &#x03B1;, Fleiss &#x03BA;, Jaccard similarity, and semantic similarity measures. Evaluation also incorporated coding density, consistency, accuracy, sensitivity, specificity, precision, recall, <italic>F</italic><sub>1</sub>-score, and overall accuracy, as well as qualitative assessments of thematic consistency, content overlap, and expert review (<xref ref-type="table" rid="table3">Table 3</xref> and Appendix IV in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>). Time efficiency was also frequently evaluated, with several studies reporting substantial reductions in analytic time relative to human-led approaches. Despite strong performance on many descriptive and structured coding tasks, human oversight remained necessary to verify outputs, identify hallucinations, assess interpretive depth, and ensure methodological rigor.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Performance of generative AI (GAI) relative to human qualitative analysis<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Analytic task</td><td align="left" valign="bottom">Typical agreement with human</td><td align="left" valign="bottom">Common metrics used</td><td align="left" valign="bottom">Noted limitations</td><td align="left" valign="bottom">References</td></tr></thead><tbody><tr><td align="left" valign="top">Inductive thematic analysis</td><td align="left" valign="top">Across evaluated models (primarily ChatGPT variants, with additional evidence from Claude, DeepSeek, Gemma, and Llama), agreement with human thematic analysis was generally high, reaching approximately 80% in 1 study. Several models reliably identified core themes and extracted meaningful insights from large health-related datasets. Some studies reported that GAI-generated subthemes added complementary depth and that hallucinations were infrequent, whereas accuracy and comprehensiveness declined with longer texts and nuanced or divergent themes. Overall, evaluated models demonstrated value for theme identification, but inductive thematic analysis still requires human initiation, guidance, and oversight.</td><td align="left" valign="top">Coding quality (density, accuracy, and consistency), agreement and overlap (theme or subtheme alignment, conceptual overlap, and hallucinations), efficiency (analysis time), and evaluation scores, quantitative ratings across 6 quality dimensions (1-7) plus qualitative assessor feedback.</td><td align="left" valign="top">Human oversight was recommended in 5 of 6 studies. Limited depth, contextual understanding, or latent interpretation was reported in 4 of 6 studies. Performance depended on prompting strategies in 3 of 6 studies (50%), while variability across models or repeated runs was noted in 2 of 6 studies. Additional limitations included misclassifications or overinterpretation (2/6), reduced performance with longer texts (1/6), dependence on predefined themes (1/6), and limited theory-driven insight. Overall, although several ChatGPT, Claude, DeepSeek, Gemma, and Llama models demonstrated useful inductive thematic analysis capabilities; however, none consistently matched the depth and contextual richness of expert human analysis.</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref26">26</xref>,<xref ref-type="bibr" rid="ref37">37</xref>-<xref ref-type="bibr" rid="ref40">40</xref>]</td></tr><tr><td align="left" valign="top">Deductive thematic analysis</td><td align="left" valign="top">Across evaluated models (primarily ChatGPT-3.5, ChatGPT-4, Bard, and Mixtral), GAI-generated themes were generally coherent, theory-aligned, and broadly consistent with human analyses, while substantially reducing analysis time. Agreement with human themes was higher for inductive analysis (~71% in 1 study) than deductive analysis (50%&#x2010;58% in 1 study), with overall human-AI coding agreement rated fair to moderate. Although performance varied across models and analytic approaches, ChatGPT-4, Bard, and Mixtral demonstrated potential as efficient complements to human-led qualitative analysis, particularly when integrated into hybrid workflows with researcher oversight.</td><td align="left" valign="top">Theme alignment and coherence with human codes, theoretical fit, theme consistency (% matched), intercoder agreement, and time efficiency; plus code frequencies, participant coverage, cross-checks (AI vs human), correctness flags, and similarity or readability scores (eg, ROUGE<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup>/BERTScore<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup> and Flesch-Kincaid).</td><td align="left" valign="top">Human oversight or hybrid human-AI workflows were recommended in 3 of 4 studies, primarily to address limitations in contextual interpretation and ensure data integrity. Concerns regarding bias, hallucinations, unsupported findings, or privacy risks were also reported in 3 of 4 studies. Reduced depth, contextual understanding, or nuanced interpretation was also identified in 3 of 4 studies. Performance differences between evaluated models were noted across studies, suggesting that limitations were not uniform across all LLMs<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup>. Lower coding reliability and coding organization limitations were reported less frequently (1/4).</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref42">42</xref>]</td></tr><tr><td align="left" valign="top">Reflexive thematic analysis</td><td align="left" valign="top">Across the evaluated models (ChatGPT-3.5 and Mistral-7B), GAI-generated themes were broadly similar to those generated by experienced researchers and were capable of generating usable codebooks and preliminary thematic structures. ChatGPT-3.5 generated plausible thematic categories and theoretical interpretations, whereas Mistral-7B was more limited to surface-level summaries and keyword-based coding. Overall, tightly controlled prompting improved performance, but both models remained less interpretive and reflexive than expert human analysis.</td><td align="left" valign="top">Content coverage of key topics, accuracy of quotes and citations (including detection of fabricated quotes), and correctness of theoretical alignment. No numerical reliability metrics were reported.</td><td align="left" valign="top">Human oversight and critical verification were recommended in both studies. Both studies reported fabricated quotes and theory misapplication or forced theoretical interpretations, although these issues varied in severity across the evaluated models. One study identified surface-level coding, missed irony, language-sensitivity issues, and increased analytic workload due to additional verification requirements. Overall, neither ChatGPT-3.5 nor Mistral-7B replaced the reflexive, interpretive role of human researchers, despite providing useful support for triangulation and preliminary analysis.</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref30">30</xref>]</td></tr><tr><td align="left" valign="top">Inductive content analysis</td><td align="left" valign="top">Performance varied, ChatGPT-3.5 and ChatGPT-4 generally provided accurate, CDC<sup><xref ref-type="table-fn" rid="table3fn5">e</xref></sup>-aligned HIV and PrEP<sup><xref ref-type="table-fn" rid="table3fn6">f</xref></sup> information and, in 1 study, offered more culturally tailored and resource-specific guidance when race was specified. ChatGPT-4 conducted qualitative analysis much faster than novice human coders while producing a comparable volume of codes, categories, and themes, although human coders generated richer and more transparent analyses. Among health information tools, Bard-Gemini produced the most comprehensive but most variable responses, ChatGPT-4 generated the most consistent outputs, and the HIV.gov chatbot provided shorter but more citation-dense responses.</td><td align="left" valign="top">Comparison of answer differences by attitude and identity, counts of codes, categories, and themes, total coding time, and expert-assessed qualitative criteria (depth, contextual richness, and transparency).</td><td align="left" valign="top">Human oversight, expert review, or ongoing monitoring was recommended in all studies (4/4). Concerns regarding bias (3/4), transparency, explainability, or citation practices (3/4), and limited depth or contextual richness (2/4) were frequently reported. Hallucinated, fabricated, or inaccurate outputs were noted in 2 of 4 studies. One study found that outputs varied according to perceived user identity, creating opportunities for tailored messaging but also risks of unintended bias, while another (1/4) reported blurred distinctions between codes, categories, and themes. Overall, limitations differed across evaluated models, with ChatGPT, Bard-Gemini, and the HIV.gov chatbot exhibiting distinct strengths and weaknesses.</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref43">43</xref>-<xref ref-type="bibr" rid="ref45">45</xref>]</td></tr><tr><td align="left" valign="top">Deductive content analysis</td><td align="left" valign="top">Across evaluated systems (ChatGPT-3.5, ChatGPT-4, Copilot, and Llama 3.1), GAI-assisted coding generally showed moderate to high agreement with human coders (&#x03BA;&#x2248;0.7&#x2010;0.96). ChatGPT closely replicated human annotations in structured tasks such as adverse-event detection, while Llama 3.1 achieved reliability comparable to human coding in large-scale content analysis. Copilot accurately reproduced manifest content but was weaker on latent interpretation, sometimes overinterpreting when additional contextual information was provided. Overall, performance was strongest for structured, deductive coding tasks, although meaningful differences were observed across models and analytic applications.</td><td align="left" valign="top">Precision of extracted mechanisms; intercoder reliability (Fleiss &#x03BA; and Krippendorff &#x03B1;, including prevalence-adjusted &#x03BA;); sensitivity (positive identification rate); reliability on full versus held-out samples; and qualitative agreement measures, completeness of meaning units, coding accuracy, similarity of sub- and over-arching themes, consensus with manual analysis, and output length or detail.</td><td align="left" valign="top">Human oversight or collaborative review was recommended in 3 of 6 studies. Reduced performance on complex, latent, or interpretive constructs was reported in 3 of 6 studies, particularly for Copilot and Llama when addressing nuanced or higher-order concepts. Bias, training data, or generalizability concerns were identified in 4 of 6 studies, spanning ChatGPT and Llama. Hallucinations, inaccuracies, or miscoding were reported in 2 of 6 studies, as were privacy and ethical concerns. Transparency and broader validation concerns were noted in 1 of 6 studies. Overall limitations varied across GAIs, suggesting that deficiencies were not uniform across all models despite generally strong performance on structured, manifest content coding tasks.</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref46">46</xref>-<xref ref-type="bibr" rid="ref49">49</xref>]</td></tr><tr><td align="left" valign="top">Grounded theory</td><td align="left" valign="top">Across the 2 studies of ChatGPT-4 and ChatGPT-4 Turbo, GAI-generated themes were broadly comparable to those produced by expert researchers and improved coding efficiency. Agreement was strongest for frequent, descriptive themes (&#x003E;80% in 1 study), whereas performance declined for culturally or emotionally nuanced themes requiring deeper interpretation (~30% agreement in 1 study). Although overall theoretical frameworks were generally similar to those produced through manual coding, limitations in depth, contextual relevance, and coding organization remained evident. Overall, ChatGPT-4 usefully supports grounded theory coding but cannot replace human interpretive expertise.</td><td align="left" valign="top">Percent agreement, Cohen &#x03BA;, node and reference counts, coverage rates, <italic>t</italic> tests, descriptive theme agreement percentages, and theme frequency counts.</td><td align="left" valign="top">Human oversight was recommended in both ChatGPT studies. Both studies reported limitations in depth, contextual understanding, and interpretive richness, as well as concerns regarding potential bias. One study identified difficulty with culturally grounded interpretation, while another reported hallucination and data privacy risks. Overall, the available evidence suggests that ChatGPT is useful for identifying surface-level themes and improving coding efficiency but cannot replace human expertise for nuanced theoretical and culturally embedded interpretation.</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref41">41</xref>]</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>ChatGPT (including GPT-3.5, GPT-4, GPT-4o, and GPT-4-Turbo) was the most frequently evaluated GAI system. Evidence for Copilot, Bard-Gemini, Claude, Llama, DeepSeek, Mistral, and other models was more limited. Consequently, findings summarized as &#x201C;GAIs&#x201D; reflect the available evidence base and should not be interpreted as evidence that all GAI systems perform similarly across qualitative analytic tasks.</p></fn><fn id="table3fn2"><p><sup>b</sup>ROUGE: Recall-Oriented Understudy for Gisting Evaluation.</p></fn><fn id="table3fn3"><p><sup>c</sup>BERTscore: Bidirectional Encoder Representations from Transformers Score.</p></fn><fn id="table3fn4"><p><sup>d</sup>LLM: large language model.</p></fn><fn id="table3fn5"><p><sup>e</sup>CDC: Centers for Disease Control.</p></fn><fn id="table3fn6"><p><sup>f</sup>PrEP: pre-exposure prophylaxis.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-5"><title>Performance of GAI Relative to Human-Led Qualitative Analysis</title><p>Across methodological approaches, GAI tools demonstrated their strongest performance in inductive thematic and content analysis, where agreement with human analysts commonly approached or exceeded 80% for core themes (<xref ref-type="table" rid="table3">Table 3</xref>). GAI tools reliably identified dominant patterns and accelerated analysis timelines, often completing tasks in a fraction of the time required for human-only workflows. However, performance consistently declined with longer texts, culturally nuanced data, or analyses requiring interpretation.</p><p>For deductive and reflexive thematic analysis, GAI tools produced coherent outputs aligned with predefined frameworks but struggled with nuance, reflexivity, and interpretive depth. Agreement with human coding was lower than with inductive approaches, and researchers frequently reported supporting superficial or keyword-driven outputs, reinforcing the point that reflexive qualitative methodologies require human interpretation and analysis. In grounded theory, GAI tools showed strong agreement on descriptive coding and frequent categories but were substantively weaker on culturally embedded or emotionally complex themes. While GAI tools improved efficiency and supported triangulation, they did not consistently generate theoretically robust core categories or relational structures, emphasizing that theory building remains a human-led intellectual task (<xref ref-type="table" rid="table3">Table 3</xref>). Across these 3 most common qualitative approaches (thematic analysis, content analysis, and grounded theory), performance variability was influenced by prompting strategies, model version, and the presence of predefined frameworks. Human oversight was universally identified as essential for ensuring interpretive accuracy, methodological rigor, and ethical compliance.</p></sec><sec id="s3-6"><title>General Findings Across Multiple Qualitative Approaches and GAI Applications</title><sec id="s3-6-1"><title>Overview</title><p>There were many qualitative approaches (eg, thematic analysis, content analysis, grounded theory, constant comparison, and framework analysis) used across the included studies, as well as many GAI models (eg, ChatGPT, Gemini, Copilot, and Claude). Below are general observations across all qualitative approaches, as well as differences across GAI models.</p></sec><sec id="s3-6-2"><title>Inductive Thematic Analysis</title><p>Across studies using inductive thematic analysis, findings were both consistent and divergent. Some evidence shows limitations: ChatGPT-4 and OpenAI o1-preview underperformed on accuracy and comprehensiveness as document length increased, likely due to reasoning constraints [<xref ref-type="bibr" rid="ref37">37</xref>]. In contrast, other research found strong performance. ChatGPT-4 achieved 85% expert-verified accuracy and high semantic alignment (0.795) when coding maternity care interviews, reducing coding time by 81% [<xref ref-type="bibr" rid="ref28">28</xref>]. ChatGPT-4o also reproduced human themes from large free-text datasets, with keywords matching NVivo outputs, although human reviewers identified errors such as fabricated quotes, paraphrasing, and merged themes, underscoring the need for careful validation, particularly for latent meanings [<xref ref-type="bibr" rid="ref33">33</xref>]. Older models, such as ChatGPT-3.5, still achieved 80% agreement with human coders, although they occasionally misclassified divergent subthemes due to overgeneralization or overfitting, reinforcing the need for human oversight [<xref ref-type="bibr" rid="ref23">23</xref>].</p><p>Comparative studies show similar patterns. For inductive classification of social media data, ChatGPT-4o with 2-shot prompting outperformed DeepSeek V3, Gemma 3, Llama 3, and ChatGPT-3.5 across all metrics, although researchers emphasized the need for human-generated initial themes to guide GAI integration [<xref ref-type="bibr" rid="ref39">39</xref>]. Another study comparing human and GAI analysis of interviews on asthma-related medication management needs found that Gemini, Copilot, and ChatGPT produced highly overlapping concepts with human NVivo coding, identifying the same 4 support domains but also misinterpreted nuances and generated themes absent from human analysis [<xref ref-type="bibr" rid="ref34">34</xref>]. Similarly, Deiner et al [<xref ref-type="bibr" rid="ref40">40</xref>] reported that ChatGPT-4 and Claude-2 consistently produced reasonable, relevant themes with low hallucination rates and strong capacity to process large social media datasets, although they still lacked the depth and consistency of expert analysts. Together, these studies suggest that GAI tools can substantially accelerate inductive thematic analysis but require rigorous human validation to ensure accuracy and interpretive depth.</p></sec><sec id="s3-6-3"><title>Deductive and Reflexive Thematic Analysis</title><p>Several studies used deductive or reflexive thematic analysis when integrating GAI tools. A locally hosted Llama-2-70B-Instruct model applying Braun and Clarke&#x2019;s 6-step reflexive thematic analysis demonstrated moderate-to-substantial similarity to human coding, suggesting that open-source models can support rigorous, secure, and scalable qualitative research [<xref ref-type="bibr" rid="ref50">50</xref>]. In contrast, Vikan et al [<xref ref-type="bibr" rid="ref24">24</xref>] found that another locally hosted model, Mistral-7B, produced only surface-level summaries, generated keyword-like codes, and failed to construct abstract or interpretive themes. The model also missed irony, fabricated quotes, introduced irrelevant theoretical links, and was highly sensitive to translation, thereby increasing researchers&#x2019; workload due to the required verification [<xref ref-type="bibr" rid="ref24">24</xref>]. The authors concluded that high-quality thematic analysis remains fundamentally human driven.</p><p>Studies using off-the-shelf models found similar patterns. ChatGPT-4 outperformed ChatGPT-3.5 and Llama-3 when summarizing complex qualitative data from an online brain tumor support forum, producing efficient and consistent summaries [<xref ref-type="bibr" rid="ref29">29</xref>]. ChatGPT-3.5-generated themes were broadly similar to those of an expert researcher in another study, although with fabricated quotes and forced theoretical interpretations, underscoring the need for human validation [<xref ref-type="bibr" rid="ref30">30</xref>]. A study that applied ChatGPT-4 to analyze LGBTQ+ patients&#x2019; positive primary care experiences showed that the model accelerated theme development, but human oversight remained essential to ensure contextual accuracy and data integrity. This hybrid workflow appears promising for improving patient-provider research applications [<xref ref-type="bibr" rid="ref42">42</xref>].</p></sec><sec id="s3-6-4"><title>Mixed Inductive and Deductive Thematic Analysis Approaches</title><p>Several studies used mixed inductive-deductive thematic approaches. Sakaguchi et al [<xref ref-type="bibr" rid="ref8">8</xref>] applied grounded theory and framework analysis to Japanese interviews using ChatGPT-4 and found more than 80% agreement with human coders for descriptive themes, but only approximately 30% agreement for culturally nuanced themes. The model was efficient for surface-level patterns but could not replace human interpretation [<xref ref-type="bibr" rid="ref8">8</xref>]. Similarly, Kon et al [<xref ref-type="bibr" rid="ref51">51</xref>] reported that ChatGPT-3.5 and ChatGPT-4 achieved 83% concordance with human analysis of community eye clinic interviews. Both models were approximately 20 times faster, with ChatGPT-4 producing fewer irrelevant subthemes [<xref ref-type="bibr" rid="ref51">51</xref>]. ChatGPT-4 also scored higher than humans on confirmability, credibility, dependability, and consistency, although humans outperformed on transferability and depth [<xref ref-type="bibr" rid="ref26">26</xref>]. In another study, ChatGPT-3.5 and Google Bard showed 71% inductive theme overlap but lower deductive consistency (50%&#x2010;58%) and only fair-to-moderate intercoder reliability; however, both completed analyses 97% faster than humans [<xref ref-type="bibr" rid="ref7">7</xref>]. Finally, a study comparing a 2-step prompting workflow using ChatGPT-3.5-turbo with a local Mixtral 7&#x00D7;8B model found that both could generate coherent inductive and deductive themes aligned with human annotations and outperformed latent semantic analysis, demonstrating their usefulness as practical complements to qualitative workflows [<xref ref-type="bibr" rid="ref27">27</xref>].</p></sec><sec id="s3-6-5"><title>Content Analysis</title><p>Content analysis was the second most common qualitative method in this review. Several studies have demonstrated that LLMs can support aspects of content analysis, particularly in generating inductive, data-driven codes. For instance, ChatGPT-3.5-Turbo performed well in inductive coding and moderately well in deductive coding using frameworks such as the theoretical domains framework, with reliability improving through iterative refinement [<xref ref-type="bibr" rid="ref48">48</xref>]. It also closely matched human annotators when detecting adverse events (AEs) across 10,000 Reddit posts, achieving more than 94% agreement for any AE and over 99% for serious AEs, suggesting strong potential for large-scale biomedical content analysis [<xref ref-type="bibr" rid="ref49">49</xref>]. Yet generalizability and model variability remain concerns. Other studies found significant shortcomings: ChatGPT-3.5 and ChatGPT-4 produced superficial and sometimes inaccurate analyses when applied to the European Resuscitation Guidelines, including hallucinated content [<xref ref-type="bibr" rid="ref47">47</xref>]. Similarly, ChatGPT-3.5 generated health information with more negative sentiment, higher reading levels, and lower DISCERN quality scores than Centers for Disease Control and Prevention (CDC) materials, underscoring the need for public literacy about AI-generated health information [<xref ref-type="bibr" rid="ref36">36</xref>]. Studies evaluating smoking cessation guidance across multiple GPT-based models found only partially reliable outputs and inconsistent references to evidence-based treatments [<xref ref-type="bibr" rid="ref52">52</xref>].</p><p>Comparisons across GAI systems showed varied performance. ChatGPT-3.5, ChatGPT-4, Google Bard, and the HIV.gov chatbot were all capable of providing accurate HIV medication safety information, although Bard was the most comprehensive and ChatGPT-4 the most consistent; the HIV.gov chatbot produced shorter but better-cited responses [<xref ref-type="bibr" rid="ref44">44</xref>]. A smaller set of studies explored manifest and latent content analysis. Copilot reproduced manifest content reasonably well but produced shorter, less nuanced interpretations, performing best when paired with a structured framework such as Graneheim and Lundman&#x2019;s [<xref ref-type="bibr" rid="ref38">38</xref>]. In another study, ChatGPT-4 rapidly completed inductive content analysis and produced a comparable number of codes and categories, but human coders developed richer, more contextually grounded themes and maintained transparent audit trails [<xref ref-type="bibr" rid="ref43">43</xref>]. Finally, a locally run Llama 3.1 model, used in an agentic workflow, effectively replicated and scaled a prior human-led content analysis, including multilingual data, although performance dropped for complex constructs such as efficacy; iterative prompting improved reliability [<xref ref-type="bibr" rid="ref46">46</xref>].</p></sec><sec id="s3-6-6"><title>Grounded Theory</title><p>Yue et al [<xref ref-type="bibr" rid="ref41">41</xref>] provided guidance for using ChatGPT in grounded theory and compared ChatGPT-4-Turbo&#x2019;s open, axial, and selective coding with human and software-assisted coding (NVivo). ChatGPT-4-Turbo produced reliable categories and theoretical structures and dramatically reduced analysis time (approximately 1 d for ChatGPT vs 3 wk for manual coding), although its coding lacked depth, contextual nuance, and strong relational connections [<xref ref-type="bibr" rid="ref41">41</xref>]. Despite slight differences in core categories, ChatGPT-4-Turbo and manual approaches yielded largely similar theoretical frameworks, highlighting both the efficiency and limits of AI-assisted grounded theory [<xref ref-type="bibr" rid="ref41">41</xref>]. Other studies used grounded theory as well, but as guidance for other qualitative methods, such as thematic analysis [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref30">30</xref>].</p></sec><sec id="s3-6-7"><title>Thematic Narrative Analysis</title><p>Using thematic narrative analysis, Chubb et al [<xref ref-type="bibr" rid="ref53">53</xref>] used ChatGPT-4o to extract themes and generate first-person vignettes from interview transcripts, substantially accelerating the synthesis while preserving participants&#x2019; voices. AI-generated themes uncovered patterns that sometimes differed from manual coding, prompting richer interpretation when triangulated with human review [<xref ref-type="bibr" rid="ref53">53</xref>]. Overall, combining human expertise with ChatGPT-4o improved efficiency without compromising ethical standards or data fidelity [<xref ref-type="bibr" rid="ref53">53</xref>].</p></sec><sec id="s3-6-8"><title>Causal Loop Diagrams</title><p>ChatGPT-4 was used to replicate a 2019 causal loop diagram of an obesity prevention intervention by extracting variables, identifying causal links, and generating feedback loops [<xref ref-type="bibr" rid="ref54">54</xref>]. It produced a richer set of feedback loops and captured new employee-level dynamics but occasionally exhibited directional errors and limited contextual nuance [<xref ref-type="bibr" rid="ref54">54</xref>]. The authors of the study concluded that GAI is a useful complement to the research but not a replacement for human qualitative analysis [<xref ref-type="bibr" rid="ref54">54</xref>].</p></sec><sec id="s3-6-9"><title>Qualitative Description</title><p>Li et al [<xref ref-type="bibr" rid="ref55">55</xref>] used an LLM called Versa (a private version of ChatGPT-4 that operates independently and stores no input data), in collaboration with human researchers, to analyze interviews on urology-related topics using a qualitative description approach. Versa consistently identified major themes with moderate agreement to human coding, but missed some nuanced subthemes, demonstrating limited depth compared to humans [<xref ref-type="bibr" rid="ref55">55</xref>]. While human analysis remained contextually richer, the model&#x2019;s reliability supports its role as a complementary tool in qualitative research [<xref ref-type="bibr" rid="ref55">55</xref>].</p></sec><sec id="s3-6-10"><title>Constant Comparison or Deductive Coding</title><p>Balt et al [<xref ref-type="bibr" rid="ref22">22</xref>] tested whether the Llama-3 (70B instruct version) could reliably perform deductive coding and summarization of psychosocial-autopsy interview data. The model achieved 84% accuracy on binary coding and produced adequate or good summaries in approximately 80% of cases; in 1.3% of cases, the summarized fragments were hallucinated or miscoded content [<xref ref-type="bibr" rid="ref22">22</xref>]. It processed 38 interviews in 8 days, compared with 4 months for the human-led analysis; however, the authors recommend a collaborative workflow in which the LLM performs initial coding, followed by expert review and refinement [<xref ref-type="bibr" rid="ref22">22</xref>]. Another study had similar conclusions, finding that LLM-assisted (ChatGPT-4o and Gemini Advanced Pro 1.5) thematic analysis with the constant comparison method quickly identified recurring patterns but often missed emotional nuance and contextual depth [<xref ref-type="bibr" rid="ref35">35</xref>]. The human-led analysis captured more complex experiences; it required much more time [<xref ref-type="bibr" rid="ref35">35</xref>]. Overall, the authors recommend a hybrid approach (human and LLM) for this type of qualitative methodology.</p></sec><sec id="s3-6-11"><title>Query-Based Analysis, Autoethnographic Case Study, and Qualitative Case Study</title><p>Morgan [<xref ref-type="bibr" rid="ref32">32</xref>] introduced a 3-step query-based analysis (QBA) workflow using ChatGPT-3.5 and ChatDOC to generate themes, subthemes, and illustrative quotes from qualitative health data. QBA produced 5 themes and subthemes and efficiently streamlined steps 3 to 5 of Braun and Clarke&#x2019;s framework while maintaining interpretive depth, although further validation is still needed. Similarly, Ferguson [<xref ref-type="bibr" rid="ref31">31</xref>] used ChatGPT-3.5 in an autoethnographic case study of comments on a newspaper article and found that the model failed to capture sentiment distinctions in satirical posts, highlighting the importance of clear prompting. To evaluate prompting strategies, Nair et al [<xref ref-type="bibr" rid="ref25">25</xref>] compared Flan-T5, ChatGPT-3, and ChatGPT-3.5 for summarizing patient forum posts. Zero-shot prompting performed reasonably well, but ChatGPT-3.5 with directional-stimulus prompting in a 3-shot setup achieved the strongest ROUGE (Recall-Oriented Understudy for Gisting Evaluation) and BERTScore (Bidirectional Encoder Representations from Transformers Score) performance, producing accurate, plausible summaries [<xref ref-type="bibr" rid="ref25">25</xref>]. The findings suggest that pretrained LLMs can generate meaningful summaries that enhance understanding of patient needs, although the study was limited by a small dataset, reliance on a single human annotator, and uncertain generalizability across models or prompting strategies [<xref ref-type="bibr" rid="ref25">25</xref>].</p></sec><sec id="s3-6-12"><title>Qualitative Comparative Analysis (QBA), Expert Appraisal</title><p>Three GAI tools (ChatGPT-4o, Gemini 1.5 Pro, and Grok) were tested for their ability to interpret and refine medical definitions of clinical obesity using structured prompts across baseline, contextual, and generative rounds [<xref ref-type="bibr" rid="ref56">56</xref>]. When provided with authoritative context (eg, the Lancet Commission Report), models produced more precise, clinically aligned definitions, suggesting GAI tools can support knowledge synthesis when paired with expert oversight [<xref ref-type="bibr" rid="ref56">56</xref>]. Bragazzi and Garbarin [<xref ref-type="bibr" rid="ref57">57</xref>] similarly found that ChatGPT-4 correctly labeled 85% of sleep-related myths as false and aligned well with expert ratings, although explanations were more general than expert technical responses. Conversely, a study evaluating ChatGPT-4o, Perplexity, Llama-3 70B, and a Retrieval-Augmented Generation&#x2013;enhanced Llama-3-70B model across 30 guideline-based cardiovascular disease nutrition questions showed that the Retrieval-Augmented Generation model provided the most reliable, guideline-adherent answers with no harmful content [<xref ref-type="bibr" rid="ref58">58</xref>]. However, its outputs were less readable because they drew directly from technical guideline language [<xref ref-type="bibr" rid="ref58">58</xref>].</p></sec></sec><sec id="s3-7"><title>Ethical Considerations of the Application of GAI-Assisted Qualitative Research</title><p>Ethical considerations were inconsistently reported across studies but emerged as a critical cross-cutting theme. As summarized in <xref ref-type="table" rid="table4">Table 4</xref>, most studies mitigated privacy risks through data deidentification and, in some cases, using enterprise- or company-hosted or locally hosted models. However, privacy risks, especially when using third-party cloud services that could retain data for training, were highlighted, and AI-specific participant consent was rarely described.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Ethical and governance approaches reported across studies<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup>.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Ethical domain</td><td align="left" valign="bottom">Mitigation strategies</td><td align="left" valign="bottom">Gaps identified</td><td align="left" valign="bottom">References</td></tr></thead><tbody><tr><td align="left" valign="top">Data privacy</td><td align="left" valign="top">Deidentification of data; enterprise (company-protected) versions of GAIs or use of local models.</td><td align="left" valign="top">Cloud retention policies unclear.</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref53">53</xref>-<xref ref-type="bibr" rid="ref55">55</xref>,<xref ref-type="bibr" rid="ref59">59</xref>]</td></tr><tr><td align="left" valign="top">Consent and IRB<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup> review</td><td align="left" valign="top">IRB approval or exemption reported.</td><td align="left" valign="top">AI-specific consent rarely described.</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref50">50</xref>-<xref ref-type="bibr" rid="ref52">52</xref>,<xref ref-type="bibr" rid="ref58">58</xref>,<xref ref-type="bibr" rid="ref60">60</xref>]</td></tr><tr><td align="left" valign="top">Hallucination risk</td><td align="left" valign="top">Human verification workflows.</td><td align="left" valign="top">Errors still detected, the challenge of AI being a &#x201C;black box&#x201D; remains.</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref54">54</xref>,<xref ref-type="bibr" rid="ref56">56</xref>,<xref ref-type="bibr" rid="ref57">57</xref>,<xref ref-type="bibr" rid="ref61">61</xref>]</td></tr><tr><td align="left" valign="top">Bias and equity</td><td align="left" valign="top">Prompt refinement; reflexive review.</td><td align="left" valign="top">Algorithmic bias seldom evaluated.</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref40">40</xref>,<xref ref-type="bibr" rid="ref45">45</xref>,<xref ref-type="bibr" rid="ref50">50</xref>,<xref ref-type="bibr" rid="ref55">55</xref>]</td></tr><tr><td align="left" valign="top">Transparency</td><td align="left" valign="top">Disclosure of GAI use.</td><td align="left" valign="top">Inconsistent reporting of prompts used or version of GAI being used (paid vs free).</td><td align="left" valign="top">[<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref53">53</xref>,<xref ref-type="bibr" rid="ref54">54</xref>]</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Mitigation strategies reflect practices reported by study authors and are not specific to any single generative AI (GAI) model.</p></fn><fn id="table4fn2"><p><sup>b</sup>IRB: institutional review board.</p></fn></table-wrap-foot></table-wrap><p>Authors stressed the importance of deidentifying data and obtaining institutional review board (IRB) or ethical approval before inputting participant-generated text into a GAI. Many studies were exempt from IRB review because the data were anonymized or publicly available. Hallucinations, fabricated quotes, and misinformation were common, prompting calls for human validation, post-processing, and clear researcher responsibility. Bias and equity were frequently evaluated explicitly, despite evidence that outputs varied by perceived user identity or prompting strategies used. Transparency practices, such as reporting model version, access type, and prompting strategies, were inconsistent, limiting reproducibility and comparability across studies (<xref ref-type="table" rid="table4">Table 4</xref>).</p></sec><sec id="s3-8"><title>Trade-Offs in Using GAI for Qualitative Health Research</title><p>Across studies, the use of GAI in qualitative health research was consistently characterized by identifiable trade-offs between efficiency, reliability, interpretive depth, and ethical considerations. <xref ref-type="table" rid="table5">Table 5</xref> synthesizes these trade-offs as explicitly reported or implied across the included studies.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Trade-offs in using generative AI (GAI) for qualitative health research<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup>.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Dimension</td><td align="left" valign="bottom">Advantages</td><td align="left" valign="bottom">Limitations</td></tr></thead><tbody><tr><td align="left" valign="top">Efficiency</td><td align="left" valign="top">Dramatically reduces time for coding, summarizing, and scaling analysis (up to 97% less time compared to human coders).</td><td align="left" valign="top">Time savings partly offset by verification, correction, and preparation prior to running the GAI.</td></tr><tr><td align="left" valign="top">Reliability</td><td align="left" valign="top">Comparable to humans for manifest or descriptive coding.</td><td align="left" valign="top">Lower reliability for latent or culturally embedded analysis.</td></tr><tr><td align="left" valign="top">Interpretive depth</td><td align="left" valign="top">Useful for identifying surface patterns.</td><td align="left" valign="top">Lacks reflexivity, emotional insight, and contextual understanding.</td></tr><tr><td align="left" valign="top">Transparency</td><td align="left" valign="top">Outputs are reproducible if prompts are held constant; audit trails are important for transparency.</td><td align="left" valign="top">Models remain as &#x201C;black boxes.&#x201D;</td></tr><tr><td align="left" valign="top">Bias and hallucination</td><td align="left" valign="top">Detectable through human review.</td><td align="left" valign="top">Training data bias of the GAIs and fabricated quotes remain risks.</td></tr><tr><td align="left" valign="top">Ethics and privacy</td><td align="left" valign="top">Local or enterprise or company models can mitigate risks.</td><td align="left" valign="top">Commercial cloud tools pose confidentiality concerns.</td></tr><tr><td align="left" valign="top">Equity and access</td><td align="left" valign="top">Enables large-scale analysis with limited staff.</td><td align="left" valign="top">Subscription costs, computer or hardware high costs, negative environmental impact.</td></tr><tr><td align="left" valign="top">Best use case</td><td align="left" valign="top">First-pass analysis, identifying surface level or manifest content, can support triangulation. Should be used in hybrid human and AI approach across different qualitative health methodologies.</td><td align="left" valign="top">It should not replace human involvement or interpretation, particularly for analyzing latent content or reflexive analysis.</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>Trade-offs represent an overall synthesis of the reviewed literature. Because most studies evaluated ChatGPT variants, findings may not apply equally to all GAI models.</p></fn></table-wrap-foot></table-wrap><p>In terms of efficiency, GAI tools dramatically reduced the time required for coding, summarization, and large-scale qualitative analysis, with several studies reporting analysis times up to 97% faster than human-led analysis. These time savings were most pronounced during first-pass coding, surface pattern identification, and processing large textual datasets. However, studies also noted that these gains were partially offset by the need for preparation (eg, prompt development and data cleaning) and post-processing tasks, including output verification, error correction, and detection of fabricated or hallucinated content. Regarding reliability, GAI tools generally performed comparably to human coders for manifest descriptive and frequently occurring content. Reliability decreased for analyses requiring latent interpretation, cultural sensitivity, or emotionally embedded meaning, with poorer performance on nuanced themes or culturally specific constructs, reinforcing that reliability varied systematically by analytic depth. Across studies, interpretive depth emerged as a key limitation. GAI tools were effective at identifying surface-level patterns and recurring topics but consistently struggled with reflexivity, emotional resonance, and context. While some AI-generated themes aligned structurally with human analyses, they often lacked the conceptual richness, reflexive richness, reflexive reasoning, and theoretical grounding present in human-led expert qualitative interpretation.</p><p>With respect to transparency, multiple studies noted that AI outputs were reproducible when prompts and inputs were held constant, and audit trails could be maintained through careful documentation. Similarly, authors highlighted the persistent opacity of model training and decision processes, limiting interpretability and raising concerns about methodological transparency. Bias and hallucination risk were reported across nearly all analytic stages. While fabricated quotes, paraphrasing errors, and biased framing were often detectable through human review, they remained recurrent risks even in studies using structured prompts and verification workflows. These findings underscore that bias and hallucination were not isolated anomalies but systematic concerns requiring ongoing oversight. Ethical and practical considerations related to privacy, equity, and access further shaped reported trade-offs. The use of local or enterprise-level models reduced confidentiality risks, whereas commercial cloud-based tools raised concerns about data retention and participant privacy. Access barriers, including subscription fees, computational requirements, and infrastructure demands, were noted as limiting the feasibility of GAI adoption for some researchers and institutions, and concerns about the environmental impact of GAI were also raised.</p><p>Taken together, the results summarized in <xref ref-type="table" rid="table5">Table 5</xref> indicate that GAI tools offer substantial practical advantages for early-stage and large-scale qualitative analysis, while acknowledging limitations in the interpretive, reflexive, and ethically sensitive aspects of qualitative health research. These trade-offs were consistent across qualitative methodologies and GAI platforms in this review.</p></sec><sec id="s3-9"><title>CASP Appraisal of the Included Studies</title><p>Overall, the body of evidence appraised using the CASP checklist demonstrates strong methodological quality across the studies, particularly in relation to the research objectives, use of qualitative methodologies, appropriate research designs, clear data collection procedures, and the presentation of findings. Most studies addressed ethical considerations, often through IRB approval, data deidentification, or justification for exemption from IRB review, and rigor in data analysis using established qualitative frameworks (eg, Braun and Clarke, 2006; and Graneheim and Lundman, 2004), and in some cases, quantified reliability metrics (eg, Fleiss &#x03BA; and Krippendorff &#x03B1;). However, several methodological limitations were identified across the studies, most notably incomplete articulation of theoretical or epistemological positioning and limited attention to reflexivity, with many studies either briefly acknowledging or omitting discussion of researcher positionality and its influence on analysis. Recruitment strategies and sampling rationales were appropriate, as many papers were methodological comparisons or exploratory; some relied on small convenience or secondary datasets, constraining transferability. Collectively, the CASP appraisal indicates that while the studies provide credible and valuable methodological insights, particularly regarding AI-assisted qualitative analysis, the literature consistently supports a hybrid human-AI approach, underscoring the continued necessity of human interpretive expertise for reflexive depth, contextual nuance, and theoretical alignment. See Appendix V in <xref ref-type="supplementary-material" rid="app6">Checklist 2</xref> for the full CASP checklist completed for each study.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><p>A clear consensus across qualitative methodologies emphasizes that a hybrid approach combining GAI with human expertise is essential for maintaining credibility, reliability, rigor, and ethical standards in qualitative health research [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref59">59</xref>]. While GAI offers efficiency, scalability, and strong pattern recognition capabilities, it cannot replace the nuanced judgment, contextual insight, or reflexive decision-making of trained qualitative researchers. Used together, GAI tools can streamline data handling and preliminary coding, while humans provide interpretive depth, methodological grounding, and ethical oversight. This hybrid model strengthens validity and ensures analyses remain rooted in lived experiences and social contexts. Despite rapid adoption, the literature reveals significant inconsistency in methodological reporting. Many studies include qualitative components but do not clearly identify the qualitative methodology guiding the analysis [<xref ref-type="bibr" rid="ref56">56</xref>-<xref ref-type="bibr" rid="ref58">58</xref>]. As qualitative approaches differ in philosophy and purpose, it is essential to name both the method (eg, grounded theory, content analysis, and thematic analysis) and the analytic approach (eg, inductive, deductive, semantic, latent, and manifest). As GAI tools become more integrated into research workflows, transparent methodological specification is critical for evaluating whether AI use enhances or undermines rigor. Conceptual ambiguity also persists, with &#x201C;LLM&#x201D; and &#x201C;GAI&#x201D; often used interchangeably, although GAI tools represent only one category of LLMs. Clear definitions will reduce confusion and improve consistency in reporting.</p><p>The rapid evolution of GAI technologies also has implications for how evidence in this field is synthesized, updated, and translated into best practices. Consequently, the surge in publications over the past 2 years shows that GAI is entering qualitative health research faster than methodological guidance can keep pace. As traditional systematic reviews are slow and often outdated by the time they are published, rapid reviews may offer a more practical way to update best practices in a fast-moving technological environment [<xref ref-type="bibr" rid="ref17">17</xref>]. Journals are increasingly receptive to rapid evidence synthesis, enabling more timely recommendations while maintaining transparent and systematic processes. Across studies, authors should report a standardized set of elements to support comparison, evaluation, and replication. Drawing on Steckler and McLeroy&#x2019;s [<xref ref-type="bibr" rid="ref62">62</xref>] external validity framework, these include recruitment and representativeness, consistency of implementation, impacts across outcomes, and information on attrition or sustainability. For GAI-specific contexts, additional reporting is needed: the model used, access mechanism (free interface, subscription, API, or local installation), version number, prompting strategy, and data handling procedures. Such disclosures improve comparability and highlight the practical constraints and opportunities associated with different GAI tools.</p><p>As GAI becomes more embedded in research practice, formal training will be essential. Integrating structured GAI instruction into public health education, especially Master of Public Health programs, would help future professionals understand both the capabilities and limits of these tools. Curricula could include modules on prompt engineering, GAI-supported qualitative and quantitative methods, and responsible use topics such as privacy, bias mitigation, reproducibility, and transparency. Faculty development is equally important, including short courses on GAI for research, ethics seminars, prompt engineering laboratories, and workshops on transparent AI-assisted writing. Together, these efforts would build a workforce skilled in using GAI while upholding ethical and scientific standards [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref63">63</xref>].</p><p>Beyond researcher training and technical competency, the reviewed studies also highlight the importance of governance frameworks for ensuring the responsible, transparent, and accountable use of GAI in qualitative health research. Across the included studies, governance concerns extended beyond technical performance and frequently encompassed privacy, transparency, accountability, bias, hallucinations, and ethical oversight. Several studies mitigated privacy risks through deidentification procedures, enterprise-protected platforms, or locally hosted models, yet concerns remained regarding cloud-based data retention, confidentiality, and the use of sensitive qualitative data in commercial models (eg, ChatGPT). These findings align with broader governance literature, which argues that GAI should be viewed as a sociotechnical system requiring oversight not only of the technology itself but also of the people, data, organizational processes, and societal contexts in which it is used. For example, Janssen [<xref ref-type="bibr" rid="ref64">64</xref>] proposed a responsible governance framework based on a complex adaptive systems perspective, emphasizing public values, data provenance, joint accountability, risk assessment, communications, and continuous oversight as key components of responsible GAI deployment. The author further argues that governance should evolve alongside GAI technology and include multiple lines of defense rather than relying on a single safeguard [<xref ref-type="bibr" rid="ref64">64</xref>].</p><p>The governance challenges identified in this review are also consistent with broader concerns raised by Taeihagh [<xref ref-type="bibr" rid="ref65">65</xref>], who noted that GAI governance must address hallucinations, opacity, bias amplification, privacy violations, misinformation, data governance, and accountability gaps through adaptive, participatory, and proactive approaches. Taeihagh [<xref ref-type="bibr" rid="ref65">65</xref>] further emphasized the importance of impact assessments, auditing, transparency requirements, stakeholder engagement, and governance mechanisms capable of responding to rapidly evolving technological capabilities. Emerging governance research in higher education demonstrates increasing institutional emphasis on approved tool lists, disclosure requirements, data protection policies, privacy safeguards, and risk management frameworks in the integration of GAI into organizational practice [<xref ref-type="bibr" rid="ref65">65</xref>]. These approaches reflect a broader shift toward balancing innovation with accountability and regulatory compliance [<xref ref-type="bibr" rid="ref65">65</xref>]. Taken together, the future integration of GAI into qualitative health research should be accompanied by formal governance structures that promote transparency, protect participant data, clarify accountability for AI-assisted outputs, and ensure that human researchers retain ultimate responsibility for interpretation, decision-making, and ethical conduct [<xref ref-type="bibr" rid="ref65">65</xref>,<xref ref-type="bibr" rid="ref66">66</xref>].</p><p>Equity concerns around access to GAI tools remain significant. Not all GAI tools are free, and many advanced models require paid subscriptions or API credits, creating cost barriers for researchers and institutions with limited resources. For example, ChatGPT&#x2019;s more powerful models are available only via subscription or paid API access, with token costs that vary by model and often increase for newer versions. Some institutional GAI tools, such as enterprise versions of Copilot, offer enhanced data protections but only through paid plans, creating disparities in access to privacy-preserving features. Likewise, locally hosted or high-parameter models require substantial hardware capacity that is not universally available, further deepening inequities. Environmental inequities compound these challenges. The physical infrastructure supporting GAI, semiconductor manufacturing, data center operations, and large-scale model training consumes vast amounts of water, energy, and raw materials. The carbon and water footprints of model training are also substantial; training GPT-3 alone required an estimated 1287 MWh of electricity and generated more than 500 tons of CO&#x2082; [<xref ref-type="bibr" rid="ref67">67</xref>]. These environmental pressures highlight the importance of transparent reporting and thoughtful policy development to avoid reinforcing existing inequities [<xref ref-type="bibr" rid="ref67">67</xref>].</p><p>Equity concerns emerged not only in access to GAI tools but also in the analytic outputs they produce. Several studies have shown that GAI tools tend to identify surface-level themes more reliably while underperforming on culturally embedded, linguistically nuanced, or emotionally complex data, raising concerns that analyses may be biased toward English-speaking populations [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref44">44</xref>]. For example, agreement between GAI and human-generated themes dropped substantially from 80% for descriptive themes to 30% for culturally (Japanese) nuanced themes [<xref ref-type="bibr" rid="ref8">8</xref>]. Concerning health literacy, health information generated from GAI, specifically ChatGPT, required a higher reading grade level, while information from the CDC was of higher quality compared to the information provided by ChatGPT [<xref ref-type="bibr" rid="ref36">36</xref>]. Importantly, only one study included in this review examined the algorithmic bias of GAI tools systematically. Chandler et al [<xref ref-type="bibr" rid="ref45">45</xref>] examined whether ChatGPT provides HIV prevention and pre-exposure prophylaxis differs based on the user&#x2019;s race and explored how this GAI could inform public health education for Black women self-educating about sexual health. ChatGPT consistently provided accurate, CDC-aligned information on HIV prevention and pre-exposure prophylaxis. When the user was described as Black, responses included more detailed, culturally attuned guidance and specific financial assistance resources, whereas generic prompts (no race specified) produced broader, less targeted overviews [<xref ref-type="bibr" rid="ref45">45</xref>]. These variations show that ChatGPT&#x2019;s outputs can shift based on perceived user identity, creating opportunities for more equitable, tailored messaging but also risks of unintended bias without careful monitoring [<xref ref-type="bibr" rid="ref45">45</xref>]. Taken together, the evidence suggests that GAI-assisted qualitative analysis risks reinforcing health inequities by amplifying widely represented narratives and diminishing marginalized experiences, underscoring the necessity of reflexive, human-led equity evidence when GAI tools are used in health-related qualitative research.</p><p>On the basis of the included studies, several practical recommendations emerge for researchers seeking to integrate GAI into qualitative health research. First, GAI should be used within a hybrid human-AI workflow, with researchers retaining responsibility for interpretation, validation, and reflexive analysis. Second, GAI outputs, including codes, themes, summaries, and quotations, should be systematically verified for inaccuracies, hallucinations, and contextual misinterpretations. Third, researchers should transparently report the GAI model, version, prompting strategy, and data handling procedures to support reproducibility and methodological rigor. Fourth, sensitive data should be protected through deidentification procedures and, where feasible, the use of enterprise-protected or locally hosted models. Finally, GAI appears most appropriate for data management, summarization, and first-pass coding tasks, whereas latent, reflexive, theory-driven, and culturally nuanced analyses should remain primarily human led.</p><p>This rapid review is not without limitations; this type of review often streamlines key steps of full systematic reviews, which can introduce bias by limiting the search scope, relying on a single reviewer, and conducting a partial review by a second reviewer. As the search was restricted to 2022 to December 2025, some recently published studies may have been missed. Articles were included only if they were published in English, which could have excluded papers in other languages, and the systematic search was conducted only in 3 major databases. This rapid review intentionally excluded studies focused solely on clinical applications of GAI, such as medical documentation, clinical decision support, surgical applications, and medical education. Although this decision allowed the review to maintain a focused examination of GAI within qualitative health research methodologies, it may have excluded relevant evidence from the broader health care AI literature. A substantial and rapidly growing body of research already examines GAI-supported clinical workflows, and future reviews should synthesize this literature separately.</p><p>In addition, the included studies used a wide range of qualitative methodologies, including thematic analysis, content analysis, and grounded theory, limiting direct comparability across studies evaluating ChatGPT variants, whereas other GAI models such as Copilot, Bard-Gemini, Claude, Llama, DeepSeek, and Mistral were evaluated less frequently, potentially limiting the generalizability of findings across the broader landscape of GAI models. Furthermore, few studies examined the influence of researcher familiarity, prompt engineering expertise, or prior experience with specific GAI models. Variations in user knowledge and prompting practices may have affected the quality and performance of GAI outputs, yet these factors were rarely measured, compared, or reported, making it difficult to distinguish model-specific capabilities from differences in user expertise. However, despite these limitations, a rapid review remains an appropriate approach for assessing the current state of evidence on a rapidly evolving technology such as GAI.</p><p>The debate over GAI&#x2019;s role in qualitative research remains active, especially in methodologies emphasizing deep reflexivity, such as reflexive thematic analysis [<xref ref-type="bibr" rid="ref6">6</xref>]. While some scholars warn that GAI could dilute interpretive depth or produce fabricated content, others argue that its widespread availability makes it unrealistic to exclude it from research practices [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref9">9</xref>]. As with the earlier introduction of computer-assisted qualitative software, the field must adapt by developing approaches that harness GAI&#x2019;s benefits while preserving the interpretive richness that defines qualitative inquiry [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref9">9</xref>]. Across the studies included in this rapid review, recurring risks, including algorithmic bias, hallucinations, erosion of interpretive nuance, and privacy breaches, can be mitigated through human-in-the-loop workflows, rigorous prompt engineering, transparent audit trails, validation against human coding, and careful use of deidentified or local data.</p><p>Importantly, this new age of GAI requires researchers to be explicit and strategic about what GAI is and what it is not used for. Appropriate uses may include data management tasks, such as transcription support, summarization, organization, and first-pass or descriptive coding, where efficiency gains do not substitute for interpretive judgment. GAI should be used with caution in inductive, latent, or theory-driven analyses and should never replace researcher-led efforts, particularly in reflexive, interpretive, or epistemologically generative phases of analysis. When GAI is used beyond descriptive phases or preparatory stages, outputs must be treated as provisional and systematically checked against the data, theory, and the researcher&#x2019;s reflexive engagement. Thus, rather than automating qualitative interpretation, GAI should function as a tool whose contributions are documented, scrutinized, and subordinated to human interpretation. Maintaining an open, reflexive, and evolving dialogue will allow the field to develop shared norms, methodological guidance, and ethical guardrails for responsible, transparent, and equitable integration of GAI in qualitative health research. This review positions GAI not only as a methodological tool but as a force that challenges how qualitative knowledge is produced, interpreted, and validated in health research.</p></sec></body><back><ack><p>The authors would like to acknowledge the Prevention Research Center at Washington University in St. Louis, the Bursky School of Public Health, for its financial support for the article processing fee. No generative AI was used to conduct the review, screen studies, extract data, synthesize findings, or draft the manuscript. During the screening process, Rayyan&#x2019;s AI-assisted duplicate detection feature was used to identify potential duplicate records. However, all 44 duplicates identified by Rayyan were manually reviewed and confirmed by a member of the research team prior to removal. The authors take full responsibility for the content of the published article.</p></ack><notes><sec><title>Funding</title><p>RDG-R was supported by grant T32 HL130357 from the National Heart, Lung, and Blood Institute, National Institutes of Health. Support for the article processing fee was provided by the Prevention Research Center at Washington University in St. Louis, the Bursky School of Public Health, and the Foundation for Barnes-Jewish Hospital.</p></sec><sec><title>Data Availability</title><p>The dataset generated or analyzed from the data extraction phase of the review is available from the corresponding author on reasonable request.</p></sec></notes><fn-group><fn fn-type="con"><p>RDG-R contributed to conceptualization, methodology, software, validation, formal analysis, investigation, resources, data curation, writing, reviewing, and editing the paper, visualization, supervision, project administration, and funding acquisition. MFS contributed to methodology, software, formal analysis, investigation, and writing, reviewing, and editing the paper. AADPDS contributed to writing, reviewing, and editing the paper and visualization. RCB contributed to resources, data curation, writing, reviewing, and editing the paper, visualization, supervision, and funding acquisition. DCP contributed to writing, review, and editing the paper and supervision. MMK contributed to writing, reviewing, and editing the paper and supervision. AAE contributed to conceptualization, methodology, investigation, resources, writing, reviewing, and editing the paper, supervision, and project administration. All authors reviewed and approved the final version of the manuscript.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AE</term><def><p>adverse event</p></def></def-item><def-item><term id="abb2">BERTscore</term><def><p>Bidirectional Encoder Representations from Transformers Score</p></def></def-item><def-item><term id="abb3">CASP</term><def><p>Critical Appraisal Skills Programme</p></def></def-item><def-item><term id="abb4">CDC</term><def><p>Centers for Disease Control</p></def></def-item><def-item><term id="abb5">GAI</term><def><p>generative AI</p></def></def-item><def-item><term id="abb6">IRB</term><def><p>Institutional Review Board</p></def></def-item><def-item><term id="abb7">JBI</term><def><p>Joanna Briggs Institute</p></def></def-item><def-item><term id="abb8">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb9">PRISMA</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses</p></def></def-item><def-item><term id="abb10">PROSPERO</term><def><p>International Prospective Register of Systematic Reviews</p></def></def-item><def-item><term id="abb11">QBA</term><def><p>query-based analysis</p></def></def-item><def-item><term id="abb12">ROUGE</term><def><p>Recall-Oriented Understudy for Gisting Evaluation</p></def></def-item><def-item><term id="abb13">SPIDER</term><def><p>sample, phenomenon of interest, design, evaluation, research type</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wheldon</surname><given-names>C</given-names> </name><name name-style="western"><surname>McKee</surname><given-names>R</given-names> </name></person-group><article-title>AI-empowered qualitative data analysis</article-title><source>CH</source><year>2025</year><volume>6</volume><issue>1</issue><pub-id pub-id-type="doi">10.15367/vd6f8w75</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Stickley</surname><given-names>T</given-names> </name><name name-style="western"><surname>O&#x2019;Caithain</surname><given-names>A</given-names> </name><name name-style="western"><surname>Homer</surname><given-names>C</given-names> </name></person-group><article-title>The value of qualitative methods to public health research, policy and practice</article-title><source>Perspect Public Health</source><year>2022</year><month>07</month><volume>142</volume><issue>4</issue><fpage>237</fpage><lpage>240</lpage><pub-id pub-id-type="doi">10.1177/17579139221083814</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Owoahene Acheampong</surname><given-names>I</given-names> </name><name name-style="western"><surname>Nyaaba</surname><given-names>M</given-names> </name></person-group><article-title>Review of qualitative research in the era of generative artificial intelligence</article-title><source>SSRN</source><pub-id pub-id-type="doi">10.2139/ssrn.4686920</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Xie</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lyu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Cai</surname><given-names>J</given-names> </name><name name-style="western"><surname>Carroll</surname><given-names>JM</given-names> </name></person-group><article-title>Harnessing the power of AI in qualitative research: exploring, using and redesigning ChatGPT</article-title><source>Computers in Human Behavior: Artificial Humans</source><year>2025</year><month>05</month><volume>4</volume><fpage>100144</fpage><pub-id pub-id-type="doi">10.1016/j.chbah.2025.100144</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Banh</surname><given-names>L</given-names> </name><name name-style="western"><surname>Strobel</surname><given-names>G</given-names> </name></person-group><article-title>Generative artificial intelligence</article-title><source>Electron Markets</source><year>2023</year><month>12</month><volume>33</volume><issue>1</issue><fpage>63</fpage><pub-id pub-id-type="doi">10.1007/s12525-023-00680-1</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jowsey</surname><given-names>T</given-names> </name><name name-style="western"><surname>Braun</surname><given-names>V</given-names> </name><name name-style="western"><surname>Clarke</surname><given-names>V</given-names> </name><name name-style="western"><surname>Lupton</surname><given-names>D</given-names> </name><name name-style="western"><surname>Fine</surname><given-names>M</given-names> </name></person-group><article-title>We reject the use of generative artificial intelligence for reflexive qualitative research</article-title><source>Qualitative Inquiry</source><year>2025</year><fpage>1</fpage><lpage>5</lpage><pub-id pub-id-type="doi">10.1177/10778004251401851</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Prescott</surname><given-names>MR</given-names> </name><name name-style="western"><surname>Yeager</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ham</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Comparing the efficacy and efficiency of human and generative AI: qualitative thematic analyses</article-title><source>JMIR AI</source><year>2024</year><month>08</month><day>2</day><volume>3</volume><fpage>e54482</fpage><pub-id pub-id-type="doi">10.2196/54482</pub-id><pub-id pub-id-type="medline">39094113</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sakaguchi</surname><given-names>K</given-names> </name><name name-style="western"><surname>Sakama</surname><given-names>R</given-names> </name><name name-style="western"><surname>Watari</surname><given-names>T</given-names> </name></person-group><article-title>Evaluating ChatGPT in qualitative thematic analysis with human researchers in the Japanese clinical context and its cultural interpretation challenges: comparative qualitative study</article-title><source>J Med Internet Res</source><year>2025</year><month>04</month><day>24</day><volume>27</volume><fpage>e71521</fpage><pub-id pub-id-type="doi">10.2196/71521</pub-id><pub-id pub-id-type="medline">40273439</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Monforte</surname><given-names>J</given-names> </name></person-group><article-title>Generative artificial intelligence and the craft of qualitative health research: observations from a techno-negative stance</article-title><source>Qual Health Res</source><year>2026</year><month>03</month><volume>36</volume><issue>2-3</issue><fpage>166</fpage><lpage>180</lpage><pub-id pub-id-type="doi">10.1177/10497323251365198</pub-id><pub-id pub-id-type="medline">40908860</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kosmyna</surname><given-names>N</given-names> </name><name name-style="western"><surname>Hauptmann</surname><given-names>E</given-names> </name><name name-style="western"><surname>Yuan</surname><given-names>YT</given-names> </name><name name-style="western"><surname>Situ</surname><given-names>J</given-names> </name><name name-style="western"><surname>Liao</surname><given-names>XH</given-names> </name><name name-style="western"><surname>Beresnitzky</surname><given-names>AV</given-names> </name><etal/></person-group><article-title>Your brain on ChatGPT: accumulation of cognitive debt when using an AI assistant for essay writing task</article-title><source>arXiv</source><year>2025</year><month>12</month><day>31</day><pub-id pub-id-type="doi">10.48550/arXiv.2506.08872v2</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gerlich</surname><given-names>M</given-names> </name></person-group><article-title>AI tools in society: impacts on cognitive offloading and the future of critical thinking</article-title><source>Societies</source><year>2025</year><volume>15</volume><issue>1</issue><fpage>6</fpage><pub-id pub-id-type="doi">10.3390/soc15010006</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dellepiane</surname><given-names>P</given-names> </name></person-group><article-title>Artificial. La nueva inteligencia y el contorno de lo humano</article-title><source>TEyET</source><issue>37</issue><fpage>e24</fpage><pub-id pub-id-type="doi">10.24215/18509959.37.e24</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Sigman</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bilinkis</surname><given-names>S</given-names> </name></person-group><source>Artificial: La Nueva Inteligencia y El Contorno de Lo Humano</source><year>2023</year><publisher-name>Debate</publisher-name><pub-id pub-id-type="other">9788419642806</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Davison</surname><given-names>RM</given-names> </name><name name-style="western"><surname>Chughtai</surname><given-names>H</given-names> </name><name name-style="western"><surname>Nielsen</surname><given-names>P</given-names> </name><etal/></person-group><article-title>The ethics of using generative AI for qualitative data analysis</article-title><source>Information Systems Journal</source><year>2024</year><month>09</month><volume>34</volume><issue>5</issue><fpage>1433</fpage><lpage>1439</lpage><pub-id pub-id-type="doi">10.1111/isj.12504</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Brereton</surname><given-names>E</given-names> </name></person-group><article-title>Colleges and universities offer faculty development for AI use in the classroom</article-title><source>EdTech Magazine</source><year>2025</year><access-date>2026-07-25</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://edtechmagazine.com/higher/article/2025/05/colleges-and-universities-offer-faculty-development-ai-use-classroom">https://edtechmagazine.com/higher/article/2025/05/colleges-and-universities-offer-faculty-development-ai-use-classroom</ext-link></comment></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Page</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>McKenzie</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Bossuyt</surname><given-names>PM</given-names> </name><etal/></person-group><article-title>The PRISMA 2020 statement: an updated guideline for reporting systematic reviews</article-title><source>BMJ</source><year>2021</year><month>03</month><day>29</day><volume>372</volume><fpage>n71</fpage><pub-id pub-id-type="doi">10.1136/bmj.n71</pub-id><pub-id pub-id-type="medline">33782057</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tricco</surname><given-names>AC</given-names> </name><name name-style="western"><surname>Khalil</surname><given-names>H</given-names> </name><name name-style="western"><surname>Holly</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Rapid reviews and the methodological rigor of evidence synthesis: a JBI position statement</article-title><source>JBI Evid Synth</source><year>2022</year><month>04</month><day>1</day><volume>20</volume><issue>4</issue><fpage>944</fpage><lpage>949</lpage><pub-id pub-id-type="doi">10.11124/JBIES-21-00371</pub-id><pub-id pub-id-type="medline">35124684</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cooke</surname><given-names>A</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>D</given-names> </name><name name-style="western"><surname>Booth</surname><given-names>A</given-names> </name></person-group><article-title>Beyond PICO: the SPIDER tool for qualitative evidence synthesis</article-title><source>Qual Health Res</source><year>2012</year><month>10</month><volume>22</volume><issue>10</issue><fpage>1435</fpage><lpage>1443</lpage><pub-id pub-id-type="doi">10.1177/1049732312452938</pub-id><pub-id pub-id-type="medline">22829486</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ouzzani</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hammady</surname><given-names>H</given-names> </name><name name-style="western"><surname>Fedorowicz</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Elmagarmid</surname><given-names>A</given-names> </name></person-group><article-title>Rayyan-a web and mobile app for systematic reviews</article-title><source>Syst Rev</source><year>2016</year><month>12</month><day>5</day><volume>5</volume><issue>1</issue><fpage>210</fpage><pub-id pub-id-type="doi">10.1186/s13643-016-0384-4</pub-id><pub-id pub-id-type="medline">27919275</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="web"><article-title>CASP Qualitative Studies Checklist</article-title><source>CASP</source><access-date>2026-02-03</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://casp-uk.net/casp-tools-checklists/qualitative-studies-checklist/">https://casp-uk.net/casp-tools-checklists/qualitative-studies-checklist/</ext-link></comment></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Long</surname><given-names>HA</given-names> </name><name name-style="western"><surname>French</surname><given-names>DP</given-names> </name><name name-style="western"><surname>Brooks</surname><given-names>JM</given-names> </name></person-group><article-title>Optimising the value of the critical appraisal skills programme (CASP) tool for quality appraisal in qualitative evidence synthesis</article-title><source>Research Methods in Medicine &#x0026; Health Sciences</source><year>2020</year><month>09</month><volume>1</volume><issue>1</issue><fpage>31</fpage><lpage>42</lpage><pub-id pub-id-type="doi">10.1177/2632084320947559</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Balt</surname><given-names>E</given-names> </name><name name-style="western"><surname>Salmi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bhulai</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Deductively coding psychosocial autopsy interview data using a few-shot learning large language model</article-title><source>Front Public Health</source><year>2025</year><volume>13</volume><fpage>1512537</fpage><pub-id pub-id-type="doi">10.3389/fpubh.2025.1512537</pub-id><pub-id pub-id-type="medline">40046117</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Castellanos</surname><given-names>A</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Gomes</surname><given-names>P</given-names> </name><name name-style="western"><surname>Vander Meer</surname><given-names>D</given-names> </name><name name-style="western"><surname>Castillo</surname><given-names>A</given-names> </name></person-group><article-title>Large language models for thematic summarization in qualitative health care research: comparative analysis of model and human performance</article-title><source>JMIR AI</source><year>2025</year><month>04</month><day>4</day><volume>4</volume><fpage>e64447</fpage><pub-id pub-id-type="doi">10.2196/64447</pub-id><pub-id pub-id-type="medline">40611510</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vikan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Aryan</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kannel&#x00F8;nning</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Riegler</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Danielsen</surname><given-names>SO</given-names> </name></person-group><article-title>Reflecting on LLM support in reflexive thematic analysis: an exploratory study</article-title><source>Qual Health Res</source><year>2026</year><month>03</month><volume>36</volume><issue>2-3</issue><fpage>191</fpage><lpage>205</lpage><pub-id pub-id-type="doi">10.1177/10497323251365211</pub-id><pub-id pub-id-type="medline">40916991</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nair</surname><given-names>RAS</given-names> </name><name name-style="western"><surname>Hartung</surname><given-names>M</given-names> </name><name name-style="western"><surname>Heinisch</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Summarizing online patient conversations using generative language models: experimental and comparative study</article-title><source>JMIR Med Inform</source><year>2025</year><month>04</month><day>14</day><volume>13</volume><fpage>e62909</fpage><pub-id pub-id-type="doi">10.2196/62909</pub-id><pub-id pub-id-type="medline">40228244</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kondo</surname><given-names>T</given-names> </name><name name-style="western"><surname>Miyachi</surname><given-names>J</given-names> </name><name name-style="western"><surname>J&#x00F6;nsson</surname><given-names>A</given-names> </name><name name-style="western"><surname>Nishigori</surname><given-names>H</given-names> </name></person-group><article-title>A mixed-methods study comparing human-led and ChatGPT-driven qualitative analysis in medical education research</article-title><source>Nagoya J Med Sci</source><year>2024</year><month>11</month><volume>86</volume><issue>4</issue><fpage>620</fpage><lpage>644</lpage><pub-id pub-id-type="doi">10.18999/nagjms.86.4.620</pub-id><pub-id pub-id-type="medline">39780933</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wosny</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hastings</surname><given-names>J</given-names> </name></person-group><article-title>Applying large language models to interpret qualitative interviews in healthcare</article-title><source>Stud Health Technol Inform</source><year>2024</year><month>08</month><day>22</day><volume>316</volume><fpage>791</fpage><lpage>795</lpage><pub-id pub-id-type="doi">10.3233/SHTI240530</pub-id><pub-id pub-id-type="medline">39176911</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Qiao</surname><given-names>S</given-names> </name><name name-style="western"><surname>Fang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>R</given-names> </name><name name-style="western"><surname>Li</surname><given-names>X</given-names> </name><name name-style="western"><surname>Kang</surname><given-names>Y</given-names> </name></person-group><article-title>Generative AI for thematic analysis in a maternal health study: coding semistructured interviews using large language models</article-title><source>Appl Psychol Health Well Being</source><year>2025</year><month>06</month><volume>17</volume><issue>3</issue><fpage>e70038</fpage><pub-id pub-id-type="doi">10.1111/aphw.70038</pub-id><pub-id pub-id-type="medline">40377231</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Muasher-Kerwin</surname><given-names>C</given-names> </name><name name-style="western"><surname>Hughes</surname><given-names>MC</given-names> </name><name name-style="western"><surname>Foster</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Al Azher</surname><given-names>I</given-names> </name><name name-style="western"><surname>Alhoori</surname><given-names>H</given-names> </name></person-group><article-title>Exploring large language models for summarizing and interpreting an online brain tumor support forum</article-title><source>Digit Health</source><year>2025</year><volume>11</volume><fpage>20552076251337345</fpage><pub-id pub-id-type="doi">10.1177/20552076251337345</pub-id><pub-id pub-id-type="medline">40297356</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wachinger</surname><given-names>J</given-names> </name><name name-style="western"><surname>B&#x00E4;rnighausen</surname><given-names>K</given-names> </name><name name-style="western"><surname>Sch&#x00E4;fer</surname><given-names>LN</given-names> </name><name name-style="western"><surname>Scott</surname><given-names>K</given-names> </name><name name-style="western"><surname>McMahon</surname><given-names>SA</given-names> </name></person-group><article-title>Prompts, pearls, imperfections: comparing ChatGPT and a human researcher in qualitative data analysis</article-title><source>Qual Health Res</source><year>2025</year><month>08</month><volume>35</volume><issue>9</issue><fpage>951</fpage><lpage>966</lpage><pub-id pub-id-type="doi">10.1177/10497323241244669</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ferguson</surname><given-names>C</given-names> </name></person-group><article-title>A researcher&#x2019;s journey to the use of AI for qualitative data analysis: findings from a test case with ChatGPT</article-title><source>Issues Educ Res</source><year>2025</year><volume>35</volume><issue>1</issue><fpage>126</fpage><lpage>141</lpage><comment><ext-link ext-link-type="uri" xlink:href="http://www.iier.org.au/iier35/ferguson.pdf">http://www.iier.org.au/iier35/ferguson.pdf</ext-link></comment></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Morgan</surname><given-names>DL</given-names> </name></person-group><article-title>Query-based analysis: a strategy for analyzing qualitative data using ChatGPT</article-title><source>Qual Health Res</source><year>2026</year><month>03</month><volume>36</volume><issue>2-3</issue><fpage>206</fpage><lpage>217</lpage><pub-id pub-id-type="doi">10.1177/10497323251321712</pub-id><pub-id pub-id-type="medline">40481623</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Goldberg</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Macis</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bounds</surname><given-names>M</given-names> </name><name name-style="western"><surname>Picazo</surname><given-names>JG</given-names> </name><name name-style="western"><surname>Nicholas</surname><given-names>LH</given-names> </name></person-group><article-title>Free-text responses in a nationally representative experimental survey about end-of-life care choices: ChatGPT-4o-assisted qualitative analytical study</article-title><source>JMIR Aging</source><year>2025</year><month>10</month><day>29</day><volume>8</volume><fpage>e76335</fpage><pub-id pub-id-type="doi">10.2196/76335</pub-id><pub-id pub-id-type="medline">41160733</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jeminiwa</surname><given-names>RN</given-names> </name><name name-style="western"><surname>Popielaski</surname><given-names>C</given-names> </name><name name-style="western"><surname>King</surname><given-names>A</given-names> </name></person-group><article-title>Exploring young adults&#x2019; experiences and beliefs in asthma medication management: pilot qualitative study comparing human and multiple AI thematic analysis</article-title><source>JMIR Form Res</source><year>2025</year><month>08</month><day>15</day><volume>9</volume><fpage>e69892</fpage><pub-id pub-id-type="doi">10.2196/69892</pub-id><pub-id pub-id-type="medline">40815807</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shanwetter Levit</surname><given-names>N</given-names> </name><name name-style="western"><surname>Saban</surname><given-names>M</given-names> </name></person-group><article-title>When investigator meets large language models: a qualitative analysis of cancer patient decision-making journeys</article-title><source>NPJ Digit Med</source><year>2025</year><month>06</month><day>5</day><volume>8</volume><issue>1</issue><fpage>336</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01747-3</pub-id><pub-id pub-id-type="medline">40473767</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Young</surname><given-names>A</given-names> </name><name name-style="western"><surname>Omosun</surname><given-names>F</given-names> </name></person-group><article-title>A comparative analysis of CDC and AI-generated health information using computer-aided text analysis</article-title><source>J Commun Healthc</source><year>2025</year><month>10</month><volume>18</volume><issue>3</issue><fpage>205</fpage><lpage>216</lpage><pub-id pub-id-type="doi">10.1080/17538068.2025.2487378</pub-id><pub-id pub-id-type="medline">40229204</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Han</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Tavasi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Can large language models be used to code text for thematic analysis? An explorative study</article-title><source>Discov Artif Intell</source><year>2025</year><volume>5</volume><issue>1</issue><fpage>171</fpage><pub-id pub-id-type="doi">10.1007/s44163-025-00441-3</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lund-Tonnesen</surname><given-names>M</given-names> </name><name name-style="western"><surname>Vahr Lauridsen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Rosenberg</surname><given-names>J</given-names> </name></person-group><article-title>Evaluating Microsoft Copilot in qualitative health research: accurate for manifest content coding but limited in latent interpretation</article-title><source>Cureus</source><year>2025</year><month>10</month><volume>17</volume><issue>10</issue><fpage>e95719</fpage><pub-id pub-id-type="doi">10.7759/cureus.95719</pub-id><pub-id pub-id-type="medline">41322837</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hairston</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Ranjan</surname><given-names>R</given-names> </name><name name-style="western"><surname>Lakamana</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Automating inductive thematic analyses of health content using large language models: a proof-of-concept study using social media data</article-title><source>JAMIA Open</source><year>2025</year><month>10</month><volume>8</volume><issue>5</issue><fpage>ooaf102</fpage><pub-id pub-id-type="doi">10.1093/jamiaopen/ooaf102</pub-id><pub-id pub-id-type="medline">40985037</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Deiner</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Honcharov</surname><given-names>V</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Mackey</surname><given-names>TK</given-names> </name><name name-style="western"><surname>Porco</surname><given-names>TC</given-names> </name><name name-style="western"><surname>Sarkar</surname><given-names>U</given-names> </name></person-group><article-title>Large language models can enable inductive thematic analysis of a social media corpus in a single prompt: human validation study</article-title><source>JMIR Infodemiology</source><year>2024</year><month>08</month><day>29</day><volume>4</volume><fpage>e59641</fpage><pub-id pub-id-type="doi">10.2196/59641</pub-id><pub-id pub-id-type="medline">39207842</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yue</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>D</given-names> </name><name name-style="western"><surname>Lv</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Hao</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cui</surname><given-names>P</given-names> </name></person-group><article-title>A practical guide and assessment on using ChatGPT to conduct grounded theory: tutorial</article-title><source>J Med Internet Res</source><year>2025</year><month>05</month><day>14</day><volume>27</volume><fpage>e70122</fpage><pub-id pub-id-type="doi">10.2196/70122</pub-id><pub-id pub-id-type="medline">40367510</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Stage</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Creamer</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Ruben</surname><given-names>MA</given-names> </name></person-group><article-title>&#x201C;Having providers who are trained and have empathy is life-saving&#x201D;: improving primary care communication through thematic analysis with ChatGPT and human expertise</article-title><source>PEC Innov</source><year>2025</year><month>06</month><volume>6</volume><fpage>100371</fpage><pub-id pub-id-type="doi">10.1016/j.pecinn.2024.100371</pub-id><pub-id pub-id-type="medline">39866208</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lockwood</surname><given-names>A</given-names> </name><name name-style="western"><surname>Newman</surname><given-names>DS</given-names> </name><name name-style="western"><surname>Mossing</surname><given-names>KW</given-names> </name><name name-style="western"><surname>Glubzinski</surname><given-names>A</given-names> </name><name name-style="western"><surname>Cohen</surname><given-names>E</given-names> </name></person-group><article-title>Human versus machine: a comparative analysis of qualitative coding by humans and ChatGPT-4</article-title><source>Sch Psychol</source><year>2026</year><month>03</month><volume>41</volume><issue>2</issue><fpage>161</fpage><lpage>172</lpage><pub-id pub-id-type="doi">10.1037/spq0000715</pub-id><pub-id pub-id-type="medline">41114954</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Beegle</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gomez</surname><given-names>LA</given-names> </name><name name-style="western"><surname>Blackard</surname><given-names>JT</given-names> </name><etal/></person-group><article-title>HIV prevention and treatment information from four artificial intelligence platforms: a thematic analysis</article-title><source>AIDS Behav</source><year>2025</year><month>11</month><volume>29</volume><issue>11</issue><fpage>3394</fpage><lpage>3403</lpage><pub-id pub-id-type="doi">10.1007/s10461-025-04786-9</pub-id><pub-id pub-id-type="medline">40481266</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chandler</surname><given-names>RD</given-names> </name><name name-style="western"><surname>Warner</surname><given-names>S</given-names> </name><name name-style="western"><surname>Aidoo-Frimpong</surname><given-names>G</given-names> </name><name name-style="western"><surname>Wells</surname><given-names>J</given-names> </name></person-group><article-title>&#x201C;What did you say, ChatGPT?&#x201D; The use of AI in Black Women&#x2019;s HIV self-education: an inductive qualitative data analysis</article-title><source>J Assoc Nurses AIDS Care</source><year>2024</year><volume>35</volume><issue>3</issue><fpage>294</fpage><lpage>302</lpage><pub-id pub-id-type="doi">10.1097/JNC.0000000000000468</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Farjam</surname><given-names>M</given-names> </name><name name-style="western"><surname>Meyer</surname><given-names>H</given-names> </name><name name-style="western"><surname>Lohkamp</surname><given-names>M</given-names> </name></person-group><article-title>A practical guide and case study on how to instruct LLMs for automated coding during content analysis</article-title><source>Soc Sci Comput Rev</source><year>2026</year><month>06</month><volume>44</volume><issue>3</issue><fpage>488</fpage><lpage>502</lpage><pub-id pub-id-type="doi">10.1177/08944393251349541</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Beck</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kuhner</surname><given-names>M</given-names> </name><name name-style="western"><surname>Haar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Daubmann</surname><given-names>A</given-names> </name><name name-style="western"><surname>Semmann</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kluge</surname><given-names>S</given-names> </name></person-group><article-title>Evaluating the accuracy and reliability of AI chatbots in disseminating the content of current resuscitation guidelines: a comparative analysis between the ERC 2021 guidelines and both ChatGPTs 3.5 and 4</article-title><source>Scand J Trauma Resusc Emerg Med</source><year>2024</year><month>09</month><day>26</day><volume>32</volume><issue>1</issue><fpage>95</fpage><pub-id pub-id-type="doi">10.1186/s13049-024-01266-2</pub-id><pub-id pub-id-type="medline">39327587</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bijker</surname><given-names>R</given-names> </name><name name-style="western"><surname>Merkouris</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Dowling</surname><given-names>NA</given-names> </name><name name-style="western"><surname>Rodda</surname><given-names>SN</given-names> </name></person-group><article-title>ChatGPT for automated qualitative research: content analysis</article-title><source>J Med Internet Res</source><year>2024</year><month>07</month><day>25</day><volume>26</volume><fpage>e59050</fpage><pub-id pub-id-type="doi">10.2196/59050</pub-id><pub-id pub-id-type="medline">39052327</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Leas</surname><given-names>EC</given-names> </name><name name-style="western"><surname>Ayers</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Desai</surname><given-names>N</given-names> </name><name name-style="western"><surname>Dredze</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hogarth</surname><given-names>M</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>DM</given-names> </name></person-group><article-title>Using large language models to support content analysis: a case study of ChatGPT for adverse event detection</article-title><source>J Med Internet Res</source><year>2024</year><month>05</month><day>2</day><volume>26</volume><fpage>e52499</fpage><pub-id pub-id-type="doi">10.2196/52499</pub-id><pub-id pub-id-type="medline">38696245</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mathis</surname><given-names>WS</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>S</given-names> </name><name name-style="western"><surname>Pratt</surname><given-names>N</given-names> </name><name name-style="western"><surname>Weleff</surname><given-names>J</given-names> </name><name name-style="western"><surname>De Paoli</surname><given-names>S</given-names> </name></person-group><article-title>Inductive thematic analysis of healthcare qualitative interviews using open-source large language models: How does it compare to traditional methods?</article-title><source>Comput Methods Programs Biomed</source><year>2024</year><month>10</month><volume>255</volume><fpage>108356</fpage><pub-id pub-id-type="doi">10.1016/j.cmpb.2024.108356</pub-id><pub-id pub-id-type="medline">39067136</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kon</surname><given-names>MHA</given-names> </name><name name-style="western"><surname>Pereira</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Molina</surname><given-names>JADC</given-names> </name><name name-style="western"><surname>Yip</surname><given-names>VCH</given-names> </name><name name-style="western"><surname>Abisheganaden</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Yip</surname><given-names>W</given-names> </name></person-group><article-title>Unravelling ChatGPT&#x2019;s potential in summarising qualitative in-depth interviews</article-title><source>Eye (Lond)</source><year>2025</year><month>02</month><volume>39</volume><issue>2</issue><fpage>354</fpage><lpage>358</lpage><pub-id pub-id-type="doi">10.1038/s41433-024-03419-0</pub-id><pub-id pub-id-type="medline">39501005</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abroms</surname><given-names>LC</given-names> </name><name name-style="western"><surname>Yousefi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wysota</surname><given-names>CN</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>TC</given-names> </name><name name-style="western"><surname>Broniatowski</surname><given-names>DA</given-names> </name></person-group><article-title>Assessing the adherence of ChatGPT chatbots to public health guidelines for smoking cessation: content analysis</article-title><source>J Med Internet Res</source><year>2025</year><month>01</month><day>30</day><volume>27</volume><fpage>e66896</fpage><pub-id pub-id-type="doi">10.2196/66896</pub-id><pub-id pub-id-type="medline">39883917</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chubb</surname><given-names>LA</given-names> </name><name name-style="western"><surname>Jackson</surname><given-names>S</given-names> </name><name name-style="western"><surname>Naseer</surname><given-names>B</given-names> </name><name name-style="western"><surname>Matthews</surname><given-names>M</given-names> </name></person-group><article-title>To leave or stay? Influences on early exit and completion in a New Zealand residential drug rehabilitation service</article-title><source>Qual Health Res</source><year>2026</year><month>03</month><volume>36</volume><issue>2-3</issue><fpage>231</fpage><lpage>246</lpage><pub-id pub-id-type="doi">10.1177/10497323251367177</pub-id><pub-id pub-id-type="medline">41069063</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jalali</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Akhavan</surname><given-names>A</given-names> </name></person-group><article-title>Integrating AI language models in qualitative research: replicating interview data analysis with ChatGPT</article-title><source>Syst Dyn Rev</source><year>2024</year><volume>40</volume><issue>3</issue><fpage>e1772</fpage><pub-id pub-id-type="doi">10.1002/sdr.1772</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>KD</given-names> </name><name name-style="western"><surname>Fernandez</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Schwartz</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Comparing GPT-4 and human researchers in health care data analysis: qualitative description study</article-title><source>J Med Internet Res</source><year>2024</year><month>08</month><day>21</day><volume>26</volume><fpage>e56500</fpage><pub-id pub-id-type="doi">10.2196/56500</pub-id><pub-id pub-id-type="medline">39167785</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kermansaravi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Cohen</surname><given-names>RV</given-names> </name></person-group><article-title>Clinical obesity through the lens of context-aware large language models</article-title><source>Obes Surg</source><year>2025</year><month>12</month><volume>35</volume><issue>12</issue><fpage>5247</fpage><lpage>5255</lpage><pub-id pub-id-type="doi">10.1007/s11695-025-08341-2</pub-id><pub-id pub-id-type="medline">41160303</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bragazzi</surname><given-names>NL</given-names> </name><name name-style="western"><surname>Garbarino</surname><given-names>S</given-names> </name></person-group><article-title>Assessing the accuracy of generative conversational artificial intelligence in debunking sleep health myths: mixed methods comparative study with expert analysis</article-title><source>JMIR Form Res</source><year>2024</year><volume>8</volume><fpage>e55762</fpage><pub-id pub-id-type="doi">10.2196/55762</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Parameswaran</surname><given-names>V</given-names> </name><name name-style="western"><surname>Bernard</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bernard</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Evaluating large language models and retrieval-augmented generation enhancement for delivering guideline-adherent nutrition information for cardiovascular disease prevention: cross-sectional study</article-title><source>J Med Internet Res</source><year>2025</year><volume>27</volume><fpage>e78625</fpage><pub-id pub-id-type="doi">10.2196/78625</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Misgav</surname><given-names>K</given-names> </name><name name-style="western"><surname>Neufeld-Kroszynski</surname><given-names>G</given-names> </name><name name-style="western"><surname>Palombo</surname><given-names>M</given-names> </name><name name-style="western"><surname>Karnieli-Miller</surname><given-names>O</given-names> </name></person-group><article-title>Human analysis vs. artificial intelligence: analyzing of qualitative medical students&#x2019; narratives</article-title><source>Qual Health Res</source><year>2026</year><month>03</month><volume>36</volume><issue>2-3</issue><fpage>218</fpage><lpage>230</lpage><pub-id pub-id-type="doi">10.1177/10497323251359445</pub-id><pub-id pub-id-type="medline">40772465</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Keating</surname><given-names>C</given-names> </name><name name-style="western"><surname>Marcus</surname><given-names>SC</given-names> </name><name name-style="western"><surname>Bowden</surname><given-names>CF</given-names> </name><name name-style="western"><surname>Worsley</surname><given-names>D</given-names> </name><name name-style="western"><surname>Doupnik</surname><given-names>SK</given-names> </name></person-group><article-title>Artificial intelligence and qualitative analysis of emergency department telemental health care implementation survey</article-title><source>Telemed J E Health</source><year>2025</year><month>07</month><volume>31</volume><issue>7</issue><fpage>821</fpage><lpage>828</lpage><pub-id pub-id-type="doi">10.1089/tmj.2024.0555</pub-id><pub-id pub-id-type="medline">40129004</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>X</given-names> </name><name name-style="western"><surname>He</surname><given-names>L</given-names> </name><name name-style="western"><surname>Alanazi</surname><given-names>E</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>E</given-names> </name><name name-style="western"><surname>Goss</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gumireddy</surname><given-names>L</given-names> </name></person-group><article-title>Assessing the accuracy and explainability of using ChatGPT to evaluate the quality of health news</article-title><source>BMC Public Health</source><year>2025</year><volume>25</volume><issue>1</issue><fpage>2038</fpage><pub-id pub-id-type="doi">10.1186/s12889-025-23206-0</pub-id></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Steckler</surname><given-names>A</given-names> </name><name name-style="western"><surname>McLeroy</surname><given-names>KR</given-names> </name></person-group><article-title>The importance of external validity</article-title><source>Am J Public Health</source><year>2008</year><month>01</month><volume>98</volume><issue>1</issue><fpage>9</fpage><lpage>10</lpage><pub-id pub-id-type="doi">10.2105/AJPH.2007.126847</pub-id></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Marshall</surname><given-names>DT</given-names> </name><name name-style="western"><surname>Naff</surname><given-names>DB</given-names> </name></person-group><article-title>The ethics of using artificial intelligence in qualitative research</article-title><source>J Empir Res Hum Res Ethics</source><year>2024</year><month>07</month><volume>19</volume><issue>3</issue><fpage>92</fpage><lpage>102</lpage><pub-id pub-id-type="doi">10.1177/15562646241262659</pub-id><pub-id pub-id-type="medline">38881315</pub-id></nlm-citation></ref><ref id="ref64"><label>64</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Janssen</surname><given-names>M</given-names> </name></person-group><article-title>Responsible governance of generative AI: conceptualizing GenAI as complex adaptive systems</article-title><source>Policy and Society</source><year>2025</year><month>01</month><day>4</day><volume>44</volume><issue>1</issue><fpage>38</fpage><lpage>51</lpage><pub-id pub-id-type="doi">10.1093/polsoc/puae040</pub-id></nlm-citation></ref><ref id="ref65"><label>65</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Taeihagh</surname><given-names>A</given-names> </name></person-group><article-title>Governance of generative AI</article-title><source>Policy and Society</source><year>2025</year><month>01</month><day>4</day><volume>44</volume><issue>1</issue><fpage>1</fpage><lpage>22</lpage><pub-id pub-id-type="doi">10.1093/polsoc/puaf001</pub-id></nlm-citation></ref><ref id="ref66"><label>66</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>LaFrance</surname><given-names>J</given-names> </name></person-group><article-title>Governing generative artificial intelligence: institutional policies and guidelines at America&#x2019;s flagship universities</article-title><source>Educational Policy</source><year>2026</year><pub-id pub-id-type="doi">10.1177/08959048261450995</pub-id></nlm-citation></ref><ref id="ref67"><label>67</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hosseini</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>P</given-names> </name><name name-style="western"><surname>Vivas-Valencia</surname><given-names>C</given-names> </name></person-group><article-title>A social-environmental impact perspective of generative artificial intelligence</article-title><source>Environmental Science and Ecotechnology</source><year>2025</year><month>01</month><volume>23</volume><fpage>100520</fpage><pub-id pub-id-type="doi">10.1016/j.ese.2024.100520</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Boolean logic syntax using all keyword combinations.</p><media xlink:href="jmir_v28i1e98551_app1.docx" xlink:title="DOCX File, 19 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Full search strategy.</p><media xlink:href="jmir_v28i1e98551_app2.docx" xlink:title="DOCX File, 22 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Data extraction variables.</p><media xlink:href="jmir_v28i1e98551_app3.docx" xlink:title="DOCX File, 14 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Main findings of generative AI application, reliability, evaluation metrics, and ethics.</p><media xlink:href="jmir_v28i1e98551_app4.docx" xlink:title="DOCX File, 72 KB"/></supplementary-material><supplementary-material id="app5"><label>Checklist 1</label><p>PRISMA checklist.</p><media xlink:href="jmir_v28i1e98551_app5.pdf" xlink:title="PDF File, 84 KB"/></supplementary-material><supplementary-material id="app6"><label>Checklist 2</label><p>CASP checklist.</p><media xlink:href="jmir_v28i1e98551_app6.docx" xlink:title="DOCX File, 35 KB"/></supplementary-material></app-group></back></article>