<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.0" xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">JMIR</journal-id>
      <journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id>
      <journal-title>Journal of Medical Internet Research</journal-title>
      <issn pub-type="epub">1438-8871</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v28i1e90872</article-id>
      <article-id pub-id-type="pmid"/>
      <article-id pub-id-type="doi">10.2196/90872</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Original Paper</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Original Paper</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Automated Health Care Thematic Analysis Using a Multiagent Large Language Model: Algorithm Development and Evaluation Study</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>Steenstra</surname>
            <given-names>Ivan</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Parameswaran</surname>
            <given-names>Vijaya</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Wang</surname>
            <given-names>Xiaomeng</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author">
          <name name-style="western">
            <surname>Xu</surname>
            <given-names>Qidi</given-names>
          </name>
          <degrees>MS</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0006-6230-2719</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author">
          <name name-style="western">
            <surname>Amjad</surname>
            <given-names>Nuzha</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0005-1821-2807</ext-link>
        </contrib>
        <contrib id="contrib3" contrib-type="author">
          <name name-style="western">
            <surname>Giles</surname>
            <given-names>Grace</given-names>
          </name>
          <degrees>BS</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0008-0824-0128</ext-link>
        </contrib>
        <contrib id="contrib4" contrib-type="author">
          <name name-style="western">
            <surname>Cumming</surname>
            <given-names>Alexa</given-names>
          </name>
          <degrees>BS</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-9284-6182</ext-link>
        </contrib>
        <contrib id="contrib5" contrib-type="author">
          <name name-style="western">
            <surname>Hermesky</surname>
            <given-names>De'angelo</given-names>
          </name>
          <degrees>BS</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0006-2697-1261</ext-link>
        </contrib>
        <contrib id="contrib6" contrib-type="author">
          <name name-style="western">
            <surname>Wen</surname>
            <given-names>Alexander</given-names>
          </name>
          <degrees>BS</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0004-4681-3450</ext-link>
        </contrib>
        <contrib id="contrib7" contrib-type="author">
          <name name-style="western">
            <surname>Kwak</surname>
            <given-names>Min Ji</given-names>
          </name>
          <degrees>DrPH</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-2778-3984</ext-link>
        </contrib>
        <contrib id="contrib8" contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Kim</surname>
            <given-names>Yejin</given-names>
          </name>
          <degrees>PhD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <address>
            <institution/>
            <institution>The University of Texas Health Science Center at Houston</institution>
            <addr-line>7000 Fannin St</addr-line>
            <addr-line>Houston, TX, 77030</addr-line>
            <country>United States</country>
            <phone>1 7135003998</phone>
            <email>yejin.kim@uth.tmc.edu</email>
          </address>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-7815-6310</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff1">
        <label>1</label>
        <institution>The University of Texas Health Science Center at Houston</institution>
        <addr-line>Houston, TX</addr-line>
        <country>United States</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Yejin Kim <email>yejin.kim@uth.tmc.edu</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>30</day>
        <month>9</month>
        <year>2026</year>
      </pub-date>
      <volume>28</volume>
      <elocation-id>e90872</elocation-id>
      <history>
        <date date-type="received">
          <day>6</day>
          <month>1</month>
          <year>2026</year>
        </date>
        <date date-type="rev-request">
          <day>2</day>
          <month>6</month>
          <year>2026</year>
        </date>
        <date date-type="rev-recd">
          <day>27</day>
          <month>7</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>28</day>
          <month>7</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©Qidi Xu, Nuzha Amjad, Grace Giles, Alexa Cumming, De'angelo Hermesky, Alexander Wen, Min Ji Kwak, Yejin Kim. Originally published in the Journal of Medical Internet Research (https://www.jmir.org), 30.09.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on https://www.jmir.org/, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://www.jmir.org/2026/1/e90872" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>Understanding patients’ experiences is essential for advancing patient-centered care, especially in chronic diseases that require ongoing communication. Qualitative thematic analysis is widely used to explore these experiences; however, the process remains labor-intensive, subjective, and difficult to scale.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>This study aimed to develop and evaluate Collaborative Theme Identification Agents (CoTI), a multiagent large language model framework designed to support manual thematic analysis by rapidly generating supporting excerpts, initial codes, and themes.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>CoTI consists of 3 agents: Instructor, Thematizer, and CodebookGenerator. The Instructor refines instruction prompts, the Thematizer extracts supporting excerpts and generates initial codes for each transcript, and the CodebookGenerator groups similar codes across all transcripts into a codebook with themes. We evaluated CoTI primarily using 12 transcripts of patient with heart failure, with a focus on perceptions of medication intensity. CoTI-generated outputs were compared against the reference standard developed by senior investigators. To explore human-AI interaction in thematic analyses, we further implemented CoTI in a user-facing application.</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>CoTI generated supporting excerpts, initial codes, and themes that were more similar to those of senior investigators than were the outputs of junior investigators, baseline natural language processing models, and other basic large language models. In an exploratory human-AI collaboration experiment, we found that the collaboration between CoTI and junior investigators provided only marginal gains compared to CoTI alone. A possible hypothesis was that junior investigators may overrely on CoTI and limit their independent critical thinking.</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>CoTI can improve the efficiency of thematic analysis by rapidly generating supporting excerpts, initial codes, and themes for human researchers’ review. These findings highlight CoTI’s potential as a useful tool for scalable qualitative research.</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>human-AI collaboration</kwd>
        <kwd>large language model</kwd>
        <kwd>multiagent framework</kwd>
        <kwd>qualitative research</kwd>
        <kwd>thematic analysis</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <p>Patient-centered care prioritizes understanding and integrating patients’ individual needs, values, and preferences into clinical decision-making [<xref ref-type="bibr" rid="ref1">1</xref>]. This approach is particularly important in the management of chronic diseases such as diabetes, hypertension, and heart failure, which require long-term treatment plans and frequent communication between patients and health care providers to ensure timely adjustments in response to changes in the patient’s condition [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>]. To explore how patients perceive and navigate their health experiences, qualitative research, particularly through thematic analysis of interview transcripts, has been widely used, offering rich sociocontextual understanding of complex health phenomena [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>]. For example, previous qualitative studies in heart failure have used thematic analysis to identify themes related to medication intensity and self-management capacity [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>].</p>
      <p>Although thematic analysis has been widely used to generate valuable insights in patient-centered care, it also faces several practical challenges. As a traditional qualitative approach, it typically relies on trained experts to manually review transcripts, extract supporting excerpts (referred to as <italic>clues</italic>), interpret their meanings to identify initial codes, and summarize these codes into a codebook with themes for interpretation (<xref rid="figure1" ref-type="fig">Figure 1</xref>A). This process is time-consuming, labor-intensive, and susceptible to subjective interpretation, as experts may differ in how they extract and summarize information. These limitations may introduce variability and potential bias into the findings [<xref ref-type="bibr" rid="ref8">8</xref>]. To address the challenges of manual thematic analysis, researchers have explored the use of natural language processing (NLP) to assist with thematic analysis [<xref ref-type="bibr" rid="ref8">8</xref>]. Traditional unsupervised NLP techniques, such as latent Dirichlet allocation (LDA) [<xref ref-type="bibr" rid="ref9">9</xref>], identify recurring patterns in word cooccurrence to generate clusters of keywords. These keyword lists are then interpreted by human analysts to assign themes. For example, Abram et al [<xref ref-type="bibr" rid="ref10">10</xref>] used LDA to identify themes for nurse interviews in the substance use field. However, this still required manual review of model-generated keywords to identify meaningful themes. Supervised NLP techniques, such as supervised BERTopic [<xref ref-type="bibr" rid="ref11">11</xref>], aim to identify themes that align with human predefined labels. However, thematic analysis is fundamentally an inductive process, where researchers typically begin with a small number (10-20) of transcripts to discover previously unknown themes. Because this task is to generate themes rather than apply existing ones, supervised NLP approaches are conceptually incompatible with thematic analysis. As a result, traditional NLP approaches either depend on human interpretation or struggle to adapt supervised frameworks to inductive discovery, making them less adaptable to the dynamic, context-rich narratives characteristic of qualitative health care research.</p>
      <p>Recent advances in large language models (LLMs), such as GPT-4 [<xref ref-type="bibr" rid="ref12">12</xref>], offer promising solutions to these limitations. LLMs can analyze long text and generate human-readable outputs in zero-shot or few-shot settings, only requiring instructions or a small number of labeled examples. This capability makes LLMs particularly well-suited for qualitative research contexts that lack annotated data or heavily rely on manual interpretation, effectively addressing key limitations of traditional NLP approaches. Prior studies have demonstrated the potential of LLMs in qualitative analysis. For example, Renard et al [<xref ref-type="bibr" rid="ref13">13</xref>] suggested that LLMs can uncover hidden insights from patient interview transcripts and identify dominant themes. Similarly, Mannstadt et al [<xref ref-type="bibr" rid="ref14">14</xref>] demonstrated that LLMs can rapidly identify dominant themes from patient interview transcripts, serving as a helpful complement to human analysis. Another study applied LLMs to identify themes about cancer patients’ experiences, demonstrating that LLMs perform well in capturing structural, temporal, and logistical aspects of narratives [<xref ref-type="bibr" rid="ref15">15</xref>]. However, because these instructions were broad, the outputs often defaulted to generic patterns and overlooked emotional nuance and contextual depth.</p>
      <p>To address these gaps, we developed Collaborative Theme Identification Agents (CoTI) to support qualitative thematic analysis with LLMs. Although frameworks such as Thematic-LM [<xref ref-type="bibr" rid="ref16">16</xref>], Thematic Analysis Framework Using Multiagent LLM (TAMA) [<xref ref-type="bibr" rid="ref17">17</xref>], and Auto-TA [<xref ref-type="bibr" rid="ref18">18</xref>] have pioneered the use of multiagent systems for social media and clinical interview data, our framework is specifically designed to capture the objective-specific insights often overlooked by general-purpose agents. CoTI integrates 3 specialized agents: Instructor is responsible for producing tailored instruction prompts to capture objective-specific insights, including psychosocial, emotional, and contextual dimensions that are often overlooked by broad instructions, while Thematizer and CodebookGenerator are designed to reflect the 2 key phases of the analytical workflow, with Thematizer extracting supporting excerpts (clues) and identifying initial codes for each transcript and CodebookGenerator summarizing these codes with similar meanings across all transcripts into themes, as each transcript contained its own set of codes that often overlapped conceptually but varied in wording. While CoTI is capable of operating as a fully automated system, LLMs are often used alongside human researchers in practice, tasked with reviewing, refining, or validating LLM-generated outputs. However, how human-AI collaboration can enhance thematic analysis quality remains insufficiently understood. To explore this, we embedded CoTI’s Thematizer in a user-facing application that enables real-time interaction with humans. This implementation provides an opportunity to examine whether combining LLMs with human involvement can improve the quality of thematic analysis in health care research contexts.</p>
      <p>Overall, the aim of this study was to develop and evaluate a multiagent LLM framework for improving the efficiency of qualitative thematic analysis and to examine whether human-AI collaboration can enhance the quality of thematic analysis in health care research.</p>
      <fig id="figure1" position="float">
        <label>Figure 1</label>
        <caption>
          <p>Study overview. To explore treatment burden perceptions among older adults with heart failure, our study compared the traditional thematic analysis with our proposed Collaborative Theme Identification Agents (CoTI) framework. (A) Traditional workflow: The manual method consists of 3 main steps. First, researchers conduct semistructured interviews and manually transcribe the audio recordings. Next, several senior investigators review each transcript to extract clues and identify initial codes. Finally, senior investigators generate themes based on clues and codes across all transcripts. This method is highly time-consuming and heavily reliant on expert labor. (B) CoTI workflow: the CoTI framework introduced a human-AI collaborative workflow that improved scalability while preserving human involvement. First, interviews were transcribed using Whisper, an automatic speech recognition tool. Before thematic analysis, Instructor (based on a heavyweight reasoning-oriented large language model [LLM]) iteratively improved the instruction prompts that guided thematic analysis. Once finalized, these instructions were injected into Thematizer (based on a lightweight general-purpose LLM) to extract clues, generate reasoning, and identify initial codes. Human review (eg, by junior investigators) and feedback on these AI-generated outputs is used to refine them. Outputs across all transcripts are processed through CodebookGenerator (based on a lightweight general-purpose LLM) to generate themes.</p>
        </caption>
        <graphic xlink:href="jmir_v28i1e90872_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
      </fig>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <sec>
        <title>Data Collection</title>
        <p>We used 10 interview transcripts that our research team collected for our previous study [<xref ref-type="bibr" rid="ref6">6</xref>] and conducted 2 more interviews for this study. Specifically, we conducted a qualitative study using one-on-one, semistructured interviews with older adults (age ≥65 years) who were hospitalized in the acute cardiac care units at Memorial Hermann Hospital at Texas Medical Center and excluded patients diagnosed with heart failure for the first time during the hospitalization, those unable to respond appropriately due to mental status changes, or those who declined to participate [<xref ref-type="bibr" rid="ref6">6</xref>].</p>
        <p>The interview guide focused on four key questions: (1) participants’ perceptions of their heart medication intensity, (2) situations in which they would feel the medications excessive, (3) factors that would make medication management easier, and (4) their overall issues in medication management [<xref ref-type="bibr" rid="ref6">6</xref>]. After conducting in-person interviews with each participant in the hospital, we transcribed the audio recordings via both professional transcription and OpenAI’s Whisper model (small size) [<xref ref-type="bibr" rid="ref19">19</xref>]. Transcripts were reviewed by interviewers to ensure accuracy.</p>
      </sec>
      <sec>
        <title>Thematic Analysis Setting</title>
        <p>We used 4 different settings to perform thematic analysis. First, 2 senior investigators independently extracted clues and identified initial codes for each transcript and developed the codebook with themes using inductive and deductive thematic analysis. Two additional independent senior investigators then reviewed all transcripts and refined themes [<xref ref-type="bibr" rid="ref6">6</xref>]. These final themes served as the reference standard in our study. Second, 4 junior investigators (working independently from senior investigators) were assigned a subset of transcripts (Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) and independently reviewed the transcripts to identify initial codes manually. Third, our proposed CoTI framework was applied to automatically extract clues and identify initial codes for each transcript and grouped similar codes across all transcripts to generate themes. Fourth, the same junior investigators used the web-based version of CoTI (see the Human-AI Collaboration Web-Based Application section) to extract clues and identify initial codes for their assigned transcripts.</p>
      </sec>
      <sec>
        <title>CoTI Model</title>
        <sec>
          <title>Model Summary</title>
          <p>CoTI implements its multiagent design through 3 LLM agents to mimic the traditional thematic analysis phases proposed by Braun and Clarke [<xref ref-type="bibr" rid="ref20">20</xref>] in 2006 (<xref rid="figure1" ref-type="fig">Figure 1</xref>). Specifically, CoTI operationalizes phases that are relatively structured and reproducible, including familiarization, coding, and theme review [<xref ref-type="bibr" rid="ref20">20</xref>]. We adopted a multiagent design because these steps involve distinct analytical functions. Separating these functions improves transparency and allows intermediate outputs to be reviewed more easily than a single LLM. Within this design, we implemented Instructor using the QwQ-32B (Alibaba Cloud) reasoning model [<xref ref-type="bibr" rid="ref21">21</xref>] to generate high-quality, tailored instruction prompts that guide Thematizer’s analysis toward capturing contextual depth, thereby supporting the familiarization phase. Thematizer was built on the GPT-4o-mini (OpenAI) model [<xref ref-type="bibr" rid="ref22">22</xref>] and extracts clues, generates reasoning, and identifies initial codes for each transcript, thereby reproducing the coding phase of the traditional thematic analysis while also allowing fast and efficient collaboration with humans to refine its outputs. CodebookGenerator, also based on the GPT-4o-mini model, summarizes similar codes across all transcripts into themes, supporting the theme reviewing phase.</p>
          <p>This study used the GPT-4o-mini model, released on July 18, 2024 [<xref ref-type="bibr" rid="ref22">22</xref>], and the QwQ-32B model, released on March 5, 2025 [<xref ref-type="bibr" rid="ref21">21</xref>]. Our research team previously published a paper about patients’ perceptions of heart failure medications on February 4, 2025 [<xref ref-type="bibr" rid="ref6">6</xref>]. Although there was a slight temporal overlap between the release of QwQ-32B and our earlier publication, we believe that all model development and data analyses had been completed prior to the release of QwQ-32B. Therefore, there was no possibility of data leakage or model memory.</p>
        </sec>
        <sec>
          <title>Instruction Generation Phase</title>
          <sec>
            <title>Overview</title>
            <p>In order to obtain high-quality instruction prompts for guiding thematic analysis while minimizing expert labor, we implemented an iterative instruction refinement process using Instructor. It began with the random selection of several transcripts, which were submitted to a reasoning model to identify initial codes. These AI-generated codes were not treated as the final codes for the transcripts but served as provisional references to guide instruction prompt development. Instructor took these AI-generated codes as inputs and progressively produced the refined clue and reasoning instruction prompts through 4 interconnected stages: clue instruction, reasoning instruction, evaluation, and optimization. Each stage built upon the previous one, ensuring a systematic progression toward high-quality clue and reasoning instruction prompts (<xref rid="figure2" ref-type="fig">Figure 2</xref>). Importantly, no senior investigator-derived reference standard was provided during this process. The optimization procedure relied solely on the study objective, transcript content, and the provisional AI-generated codes.</p>
            <fig id="figure2" position="float">
              <label>Figure 2</label>
              <caption>
                <p>Workflow of instructor. Instructor is a multiagent framework designed to iteratively refine instruction prompts for Thematizer to perform thematic analysis. The process begins with a reasoning model (QwQ-32B) generating initial codes from several randomly selected transcripts. Instructor then launches a 4-agent optimization cycle: clue-large language model (LLM), reasoning-LLM, evaluation-LLM, and optimization-LLM. First, clue-LLM, using the current clue prompt, takes each transcript and its corresponding AI-generated codes as input to generate clues for each code. Then, reasoning-LLM, guided by the current reasoning prompt, receives these clues and codes to generate reasoning statements. Next, the evaluation-LLM takes every (clue, reasoning, and code) triplet to produce feedback outlining issues with the current prompts and suggestions for improvement. Finally, the optimization-LLM takes this feedback with the current clue and reasoning prompts to generate updated versions. These updated prompts replace the previous ones, initiating the next iteration of refinement.</p>
              </caption>
              <graphic xlink:href="jmir_v28i1e90872_fig2.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
            </fig>
          </sec>
          <sec>
            <title>Initial Code Discovery</title>
            <p>To initiate the refinement process, we used the QwQ-32B reasoning model to identify initial codes in 2 randomly selected transcripts. The model was prompted to identify codes related to patients’ perceptions of the treatment burden or intensity of heart failure medications (see prompts in Textbox S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p>
          </sec>
          <sec>
            <title>Clue Instruction</title>
            <p>The first stage of refinement focused on extracting clues to support the interpretation of identified codes. For each training transcript paired with its corresponding AI-generated codes, we prompted an LLM agent (referred to as clue-LLM, implemented using the QwQ-32B reasoning model) with an initial instruction, “list clues (ie, key phrases, contextual information, semantic and emotional tones, temporal information, symptom descriptions) in the following patient-doctor dialogue that support each given identified topic.” [<xref ref-type="bibr" rid="ref23">23</xref>] (see prompts in Textbox S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). This instruction guided the model to find supporting excerpts for each code directly from the transcript.</p>
            <p>Clue-LLM was specifically instructed to find clues in the form of direct quotes from transcripts, as these preserve the exact language and context, ensuring the original meaning. Unlike summaries or interpretations, direct quotes provide original and unchanged references, reducing bias and enhancing transparency. They also serve as reliable and traceable memory units, locating specific parts of the interview.</p>
          </sec>
          <sec>
            <title>Reasoning Instruction</title>
            <p>The second stage aimed to formalize the logical relationships between the extracted clues and their associated codes through structured reasoning. Using the clues generated by clue-LLM, we prompted a second LLM agent (referred to as reasoning-LLM, implemented using the QwQ-32B reasoning model) with an initial instruction: “Based on the given clues, generate the reasoning process that supports the identified topics.” [<xref ref-type="bibr" rid="ref23">23</xref>] (see prompts in Textbox S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). This enabled the model to generate structured reasoning statements that clarified how the provided clues supported each code, thereby enhancing the clarity and explainability.</p>
          </sec>
          <sec>
            <title>Evaluation</title>
            <p>The third stage involved assessing the quality of the generated clues and reasoning. Rather than collecting feedback for each individual (clue, reasoning, and code) pair, we prompted a third LLM agent (referred to as evaluation-LLM, implemented using the QwQ-32B reasoning model) to identify common issues and offer overall suggestions for improving clue and reasoning instructions (see prompts in Textbox S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). This aggregated feedback addressed several limitations associated with individual feedback. Individual evaluations may result in inconsistent insights and overemphasize isolated patterns while overlooking systemic issues. Additionally, providing detailed feedback for each pair may increase complexity and reduce clarity in the evaluation process. By synthesizing feedback across multiple training examples, evaluation-LLM was able to provide a more comprehensive assessment, enabling consistent and scalable improvements.</p>
          </sec>
          <sec>
            <title>Optimization</title>
            <p>The final stage focused on refining the clue and reasoning instructions based on feedback provided by evaluation-LLM. During this step, a fourth LLM agent (referred to as optimization-LLM) simultaneously improved the clue and reasoning instructions to address previously identified issues and enhance their overall quality. To mitigate the impact of potentially spurious feedback from the evaluation-LLM, we prompted optimization-LLM (implemented using the QwQ-32B reasoning model) with the instruction “The feedback may be noisy, identify what is important and what is correct.” [<xref ref-type="bibr" rid="ref24">24</xref>] (see prompts in Textbox S5 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). This prompt encouraged optimization-LLM to apply critical thinking, allowing it to prioritize relevant and accurate suggestions. The overall refinement process in the preprocessing phase was iterative, involving multiple cycles of clue and reasoning generation, evaluation, and optimization.</p>
          </sec>
        </sec>
        <sec>
          <title>Thematic Analysis Phase</title>
          <sec>
            <title>Overview</title>
            <p>The thematic analysis phase aimed to apply the refined clue and reasoning instruction prompts to identify initial codes for all transcripts using Thematizer and subsequently summarize these codes across all transcripts into a codebook with themes by CodebookGenerator.</p>
          </sec>
          <sec>
            <title>Codes Identification</title>
            <p>To enable rapid code identification for a given transcript, particularly in the setting where junior investigators’ feedback is incorporated, we prompted Thematizer (implemented using the GPT-4o-mini model, which has faster inference speed compared with the QwQ-32B model used in Instructor) with the refined clue and reasoning instructions obtained from the instruction generation phase (see prompts in Textbox S6 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). To capture a broader set of candidate codes, we repeated the code identification process 3 times and took the union of all codes generated across the 3 runs (see prompts in Textbox S7 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p>
          </sec>
          <sec>
            <title>Themes Generation</title>
            <p>Because each transcript had its own set of codes after Thematizer, many of which overlapped but differed in wording, we used CodebookGenerator (implemented using the GPT-4o-mini model, for its faster inference speed) to group similar, semantically related, or duplicate codes across all transcripts into higher-level themes (see prompts in Textbox S8 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Each resulting theme included a theme name, a description, the original codes included, and representative clues, which was suitable for human review.</p>
          </sec>
          <sec>
            <title>Human-AI Collaboration Web-Based Application</title>
            <p>We converted the CoTI framework into a user-friendly, web-based application to facilitate interaction between human (eg, junior investigators) and our model (<xref rid="figure3" ref-type="fig">Figure 3</xref>). The application was built on Thematizer, which is responsible for extracting clues and identifying initial codes from individual transcripts. Because themes were generated through the analysis of all transcripts, and junior investigators in our study were each assigned only a subset of transcripts, CodebookGenerator was not converted into a human-collaboration module in our design. We selected a COVID-19 transcript as a demonstration example [<xref ref-type="bibr" rid="ref25">25</xref>]. Within the application, users (eg, junior investigators) are prompted to enter their Azure API credentials and upload a transcript. The uploaded transcript is displayed in a viewing panel on the right-hand side of the interface. Upon clicking the “Process Transcript” button, Thematizer automatically analyzes the transcript to extract clues and identify corresponding initial codes. The outputs are displayed for user review, allowing them to locate the relevant text segments in the transcript using keywords derived from the extracted clues (<xref rid="figure3" ref-type="fig">Figure 3</xref>A). If users are not satisfied with the outputs, they are offered 2 options: “Try Again,” which reprocesses the transcript without feedback, or “Provide Feedback,” which refines the model’s outputs based on the user’s input. This iterative feedback loop continues until the user indicates satisfaction by selecting “I’m Satisfied,” at which point the final results are saved (<xref rid="figure3" ref-type="fig">Figure 3</xref>B).</p>
            <fig id="figure3" position="float">
              <label>Figure 3</label>
              <caption>
                <p>Web-based application. (A) After entering the Azure OpenAI API credentials and uploading a transcript, users can view the content on the right side of the interface. (B) Once the initial model-generated response is provided, users are presented with 3 options: “I’m satisfied,” “Try again,” and “Provide feedback.” Selecting “I’m satisfied” indicates that the user accepts the response and concludes the code identification process. Choosing “Try again” prompts the model to reprocess the transcript and generate a new response. If “Provide feedback” is selected, the application displays text boxes where users can enter feedback. The model then incorporates this feedback to refine its previous response.</p>
              </caption>
              <graphic xlink:href="jmir_v28i1e90872_fig3.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
            </fig>
          </sec>
        </sec>
      </sec>
      <sec>
        <title>Evaluation Against Senior Investigators</title>
        <p>Because qualitative analysis is inherently interpretive rather than purely objective, thematic analysis cannot be considered a simple pattern-recognition task. Accordingly, the purpose of our evaluation was not to determine whether model outputs were objectively correct, but to assess how completely the model captures insights identified by senior investigators.</p>
        <p>To achieve this, we evaluated our model’s output (clues, codes, and themes) against those provided by senior investigators, which served as the reference standard. We first compared model-extracted clues with those provided by senior investigators for each transcript. To quantify similarity, we used multiple metrics, including Jaccard similarity, precision, recall, and <italic>F</italic><sub>1</sub>-score. Jaccard similarity measured the degree of overlap between model-extracted and senior-extracted clues. Precision reflected the proportion of correctly extracted clues among all extracted clues, while recall captured the proportion of reference clues that the model successfully extracted. Next, to evaluate code similarity, we calculated the cosine similarity between model-identified codes and those identified by senior investigators using embeddings generated from OpenAI’s text-embedding-3-small model, which was independent of the models used in our main framework. Specifically, for each reference code, we identified the maximum cosine similarity with any model-generated code, averaged these values across all reference codes, and then averaged them across all transcripts. Finally, we assessed theme similarity by comparing the model-generated themes with the themes generated by senior investigators, using the same embedding-based cosine similarity method applied in the code evaluation.</p>
      </sec>
      <sec>
        <title>User Perception Survey</title>
        <p>To understand junior investigators’ perceptions of CoTI’s outputs and their experience, we implemented a survey guided by the quality, understanding, expression, safety, and trust (QUEST) framework [<xref ref-type="bibr" rid="ref26">26</xref>] (see detailed questionnaire in Section B in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). We had 4 independent junior investigators (medical school students who were not involved in the original theme generation project conducted by senior investigators) participating in this phase. Each junior investigator first reviewed their assigned transcripts to identify relevant codes manually and then used the web-based version of the CoTI application to perform the same task, after which they completed the survey.</p>
      </sec>
      <sec>
        <title>Baseline Models</title>
        <p>We established 2 types of baseline models for comparison. The first type included traditional unsupervised topic modeling methods such as LDA [<xref ref-type="bibr" rid="ref9">9</xref>], Top2Vec [<xref ref-type="bibr" rid="ref27">27</xref>], and BERTopic [<xref ref-type="bibr" rid="ref28">28</xref>]. These methods were selected because they are commonly used NLP approaches for identifying latent topics in textual data. However, they are not designed to perform the full qualitative thematic analysis workflow, such as extracting supporting excerpts and identifying initial codes for each transcript. Particularly, these baseline models take all transcripts as input and typically require preprocessing steps including data cleaning, tokenization, stopword removal, and lemmatization. Each baseline model outputs a set of latent themes represented as ranked lists of high-probability keywords. Since these keyword-based representations differ from human-interpretable themes, we developed a standardized evaluation framework for comparison. Specifically, we extracted the top 6 keywords from each baseline model’s output as proxies for the generated themes. We then converted both keywords and senior investigators-generated themes into embeddings and computed cosine similarity between them to assess theme quality.</p>
        <p>The second baseline type (referred to as basic LLM) used a reasoning-oriented model (QwQ-32B) without refined instructions for clue extraction and reasoning generation. Because reasoning-oriented LLMs typically have longer inference times (GPT-4o-mini model required approximately 1 minute per transcript to complete 3 runs with aggregation, whereas a single run of QwQ-32B took approximately 3 minutes per transcript), we evaluated this baseline using a single run per transcript rather than the multiple aggregated runs used for CoTI. Unlike traditional models, this baseline was capable of generating clues, codes, and themes, allowing for direct comparison against senior investigators across all components.</p>
      </sec>
      <sec>
        <title>Ethical Considerations</title>
        <p>The interview study was conducted in accordance with the Declaration of Helsinki and approved by the institutional review board (IRB) of the University of Texas Health Science Center at Houston (HSC-MS-21-0874). This study was approved by the Committee for the Protection of Human Subjects of the University of Texas Health Science Center at Houston (protocol HSC-SBMI-13-0549).</p>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <sec>
        <title>Overview</title>
        <p>We developed CoTI, a multiagent human-AI collaborative framework designed to automate qualitative thematic analysis. Our goal was to efficiently generate high-quality outputs (clues, codes, and themes) that were similar to those produced by human experts (eg, senior investigators). We evaluated CoTI’s performance across three tasks: (1) clue extraction (an intermediate output, ie, supporting excerpts from transcripts), (2) code identification (the primary output, ie, codes derived from clues) within each transcript, and (3) theme generation (across all transcripts, summarizing all codes into a codebook with themes). Our experiments showed that clues, codes, and themes that were identified by CoTI were more similar to those of senior investigators than were the outputs of traditional NLP models, basic LLMs, or human researchers with lower levels of experience (eg, junior investigators). Moreover, collaboration between CoTI and junior investigators did not lead to outputs that were more similar to those of the senior investigator than CoTI alone.</p>
      </sec>
      <sec>
        <title>Interview Data Collection and Patient Characteristics</title>
        <p>As a case study, we conducted interviews with 12 patients with heart failure from Memorial Hermann Hospital at the Texas Medical Center in Houston, Texas, to explore their perceptions of challenges in using heart failure medications [<xref ref-type="bibr" rid="ref6">6</xref>]. Among the 12 participants, 8 (66.67%) were female, with a mean age of 74.75 (SD 8.21) years. 5 (41.67%) participants were White, and 5 (41.67%) were African American. Detailed demographic and clinical information for each participant is presented in <xref ref-type="table" rid="table1">Table 1</xref>.</p>
        <table-wrap position="float" id="table1">
          <label>Table 1</label>
          <caption>
            <p>Participant demographic and clinical characteristics.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="100"/>
            <col width="90"/>
            <col width="130"/>
            <col width="110"/>
            <col width="460"/>
            <col width="110"/>
            <thead>
              <tr valign="top">
                <td>Age (years)</td>
                <td>Sex</td>
                <td>Heart failure type</td>
                <td>Comorbid conditions, n</td>
                <td>Comorbid conditions</td>
                <td>Prescribed medications, n</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>71</td>
                <td>Female</td>
                <td>HFpEF<sup>a</sup></td>
                <td>9</td>
                <td>Atrial fibrillation, anemia of chronic disease, chronic kidney disease, diabetes mellitus, hypertension, hyperlipidemia, morbid obesity, and obstructive sleep apnea</td>
                <td>16</td>
              </tr>
              <tr valign="top">
                <td>82</td>
                <td>Female</td>
                <td>HFrEF<sup>b</sup></td>
                <td>3</td>
                <td>Breast cancer, hyperlipidemia, and hypertension</td>
                <td>20</td>
              </tr>
              <tr valign="top">
                <td>67</td>
                <td>Female</td>
                <td>HFpEF</td>
                <td>5</td>
                <td>Stroke, pulmonary embolism, atrial fibrillation, and hypertension</td>
                <td>7</td>
              </tr>
              <tr valign="top">
                <td>74</td>
                <td>Male</td>
                <td>HFpEF</td>
                <td>4</td>
                <td>Atrial fibrillation, chronic obstructive pulmonary disease, diabetes mellitus, and hypertension</td>
                <td>7</td>
              </tr>
              <tr valign="top">
                <td>69</td>
                <td>Female</td>
                <td>HFrEF</td>
                <td>6</td>
                <td>Atrial fibrillation, alcohol abuse, anemia of chronic disease, hypertension, hypothyroidism, and mild liver disease</td>
                <td>6</td>
              </tr>
              <tr valign="top">
                <td>70</td>
                <td>Female</td>
                <td>HFpEF</td>
                <td>8</td>
                <td>Coronary artery disease, chronic obstructive pulmonary disease, stroke, diabetes mellitus, hypertension, hypothyroidism, peripheral arterial disease, and obstructive sleep apnea</td>
                <td>15</td>
              </tr>
              <tr valign="top">
                <td>66</td>
                <td>Female</td>
                <td>HFpEF</td>
                <td>9</td>
                <td>Chronic obstructive pulmonary disease, coronary artery disease, diabetes mellitus, gastroesophageal reflux disease, hypertension, hyperlipidemia, hypothyroidism, morbid obesity, and obstructive sleep apnea</td>
                <td>14</td>
              </tr>
              <tr valign="top">
                <td>75</td>
                <td>Male</td>
                <td>HFrEF</td>
                <td>6</td>
                <td>Coronary artery disease, atrial fibrillation, end-stage renal disease, hypertension, hyperlipidemia, and hypothyroidism</td>
                <td>13</td>
              </tr>
              <tr valign="top">
                <td>85</td>
                <td>Male</td>
                <td>Unknown</td>
                <td>7</td>
                <td>Atrial fibrillation, diabetes mellitus, hyperlipidemia, hypertension, hypothyroidism, major depression, and peptic ulcer disease</td>
                <td>11</td>
              </tr>
              <tr valign="top">
                <td>88</td>
                <td>Female</td>
                <td>HFpEF</td>
                <td>7</td>
                <td>Coronary artery disease, chronic obstructive pulmonary disease, cerebral venous sinus thrombosis, hypertension, atrial fibrillation, chronic kidney disease, and obstructive sleep apnea</td>
                <td>14</td>
              </tr>
              <tr valign="top">
                <td>85</td>
                <td>Female</td>
                <td>HFpEF</td>
                <td>4</td>
                <td>Mitral regurgitation, hypertension, chronic kidney disease, and monoclonal gammopathy of undermined significance</td>
                <td>8</td>
              </tr>
              <tr valign="top">
                <td>76</td>
                <td>Male</td>
                <td>HFrEF</td>
                <td>5</td>
                <td>Atrial fibrillation, hypertension, chronic kidney disease, and stroke, and hyperlipidemia</td>
                <td>11</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table1fn1">
              <p><sup>a</sup>HFpEF: heart failure with preserved ejection fraction.</p>
            </fn>
            <fn id="table1fn2">
              <p><sup>b</sup>HFrEF: heart failure with reduced ejection fraction.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>CoTI Produced a Codebook Similar to the Senior Investigators’ Codebook</title>
        <p>To evaluate whether CoTI could perform thematic analysis similarly to senior investigators, both senior investigators and CoTI extracted clues, identified codes, and developed a codebook with themes, respectively. We then compared the similarity between them.</p>
        <p>We first used Instructor to refine instruction prompts that will guide Thematizer (see initial code discovery output in Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). As shown in <xref ref-type="table" rid="table2">Table 2</xref>, the instruction prompt to extract clues has evolved from an initial, general request for key phrases into a more structured instruction emphasizing direct quotes, contextual completeness, exclusive code assignment, and causal clarity. Similarly, the instruction prompt to generate reasoning advanced to require code-specific, stepwise causal chains, and precise language grounded solely in provided clues.</p>
        <p>Then, we applied Thematizer with these refined instruction prompts to extract clues and identify initial codes from all (n=12) transcripts. We calculated Jaccard similarity, precision, recall, and <italic>F</italic><sub>1</sub>-score between clues extracted by CoTI and those extracted by senior investigators to evaluate clue similarity. We also computed cosine similarity between initial codes generated by CoTI and those identified by senior investigators to measure code similarity. As shown below and in <xref ref-type="table" rid="table3">Table 3</xref>, our model produced clues and codes more similar to those of senior investigators than did the basic QwQ-32B LLM across most evaluation metrics. Specifically, relatively greater gains were observed in Jaccard score (+7.8%) and precision (+10.9%), with modest gains in <italic>F</italic><sub>1</sub>-score (+5.4%) and cosine similarity (+4.9%; 0.431 vs 0.411). The response time to extract clues and identify codes was approximately 1 minute per transcript for CoTI and 3 minutes per transcript for the basic QwQ-32B model.</p>
        <table-wrap position="float" id="table2">
          <label>Table 2</label>
          <caption>
            <p>Optimized instruction prompts by Instructor.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="160"/>
            <col width="420"/>
            <col width="420"/>
            <thead>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>To extract clues</td>
                <td>To generate reasoning</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Before optimization</td>
                <td>List clues (ie, key phrases, contextual information, semantic and emotional tones, temporal information, symptom descriptions) in the following patient-doctor dialogue that support each given identified topic.</td>
                <td>Based on the given clues, generate the reasoning process that supports the identified topics.</td>
              </tr>
              <tr valign="top">
                <td>After optimization</td>
                <td>Extract **direct quotes** from the dialogue that explicitly support each topic, ensuring:<break/>1. **Contextual Completeness &amp; Causal Links**: include specific details that clarify mechanisms or outcomes (eg, *”After doubling the dose, my BP remained at 160/100 despite the doctor’s adjustment”* instead of *”dose increase failed”*). Specify quantitative data, professional feedback, or patient-reported outcomes to strengthen causal relationships.<break/>2. **Exclusive Topic Assignment**: assign each quote to only one topic unless it explicitly addresses multiple themes *simultaneously* (eg, *”My fixed income can’t cover my 12 pills daily”* links both financial burden and polypharmacy). Avoid cross-topic bleeding (eg, *”too many pills”* for cost vs. adherence).<break/>3. **Clarity in Reuse**: for quotes used across topics (eg, religious coping statements), append contextual phrases to clarify relevance (eg, *”I put everything in the Lord’s hands [to cope with stress]”* for psychological themes vs. *”...to accept my medication burden”* for adherence topics).<break/>4. **Causal Precision**: prioritize quotes establishing explicit cause-effect chains (eg, *”The rash from the new pill made me stop taking it”* instead of *”I stopped the pill”*). Specify whether effects are patient-reported, caregiver-observed, or clinically measured.<break/>5. **Discrepancy Framing**: when perspectives conflict, frame quotes within the topic’s context (eg, *”Patient says ‘I take all meds,’ but caregiver notes ‘she skips 3 pills weekly’”* under adherence challenges).<break/>Ensure quotes are concise but include sufficient detail to avoid ambiguity and support rigorous reasoning.</td>
                <td>For each topic, construct a logical chain connecting clues to the topic by:<break/>1. **Numbered Stepwise Causality**: break down causal pathways into explicit, sequential steps (eg, *”Step 1: Eliquis caused bleeding → Step 2: Fear of overmedication → Step 3: Reduced adherence → Step 4: Uncontrolled condition”*).<break/>2. **Mechanism &amp; Behavioral Impact**: specify *how* each clue leads to outcomes, including patient behavior changes (eg, *”Step 1: High pill count → Step 2: Cognitive overload → Step 3: Missed doses → Step 4: Worsened polypharmacy burden”*; *”Step 1: Financial strain → Step 2: Delayed ER visits → Step 3: Complication escalation”*).<break/>3. **Avoid Assumptions**: explicitly map clues to outcomes using only provided data (eg, *”Step 1: Dose escalation caused nausea → Step 2: Nausea reduced medication intake → Step 3: Suboptimal BP control”* instead of implying indirect links).<break/>4. **Address Contradictions**: explain discrepancies as causal factors (eg, *”Step 1: Patient denies non-adherence → Step 2: Caregiver notes missed doses → Step 3: Conflicting narratives → Step 4: Potential for unmanaged symptoms”*).<break/>5. **Distinct Factor Differentiation**: separate overlapping effects (eg, *”Step 1: High dosage → Step 2: Nausea → Step 3: Reduced adherence”* vs. *”Step 1: Drug interactions → Step 2: Dizziness → Step 3: Fall risk”*).<break/>6. **Actionable Language**: use precise terms like *”triggers,”* *”results in,”* or *”directly causes”* to replace vague phrasing.<break/>Ensure reasoning is topic-specific, free of redundancy, and grounded solely in provided clues.</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <table-wrap position="float" id="table3">
          <label>Table 3</label>
          <caption>
            <p>Evaluation of clue similarity on the heart failure interview transcripts.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="190"/>
            <col width="210"/>
            <col width="130"/>
            <col width="0"/>
            <col width="470"/>
            <thead>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td colspan="3">Clue extraction</td>
                <td>Relative change (%)</td>
              </tr>
              <tr valign="bottom">
                <td>
                  <break/>
                </td>
                <td>Basic QwQ-32B LLM</td>
                <td>CoTI<sup>a</sup></td>
                <td colspan="2">
                  <break/>
                </td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Jaccard similarity</td>
                <td>0.374</td>
                <td>0.403</td>
                <td colspan="2">+7.75</td>
              </tr>
              <tr valign="top">
                <td>Precision</td>
                <td>0.496</td>
                <td>0.550</td>
                <td colspan="2">+10.87</td>
              </tr>
              <tr valign="top">
                <td>Recall</td>
                <td>0.635</td>
                <td>0.630</td>
                <td colspan="2">–0.79</td>
              </tr>
              <tr valign="top">
                <td><italic>F</italic><sub>1</sub>-score</td>
                <td>0.540</td>
                <td>0.569</td>
                <td colspan="2">+5.37</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table3fn1">
              <p><sup>a</sup>CoTI: Collaborative Theme Identification Agents.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <p>After generating code for each transcript, we used CodebookGenerator to summarize codes across all transcripts into a structured codebook with themes (see outputs in Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). We calculated the cosine similarity between the CoTI-generated and senior-generated themes to evaluate theme similarity. CoTI-generated themes were more similar to the senior investigator’s than those produced by traditional NLP topic modeling methods (LDA: 0.391; Top2Vec: 0.377; BERTopic: 0.266) or the basic QwQ-32B LLM (+22.24%; 0.621 vs 0.508). The codebook generation time was approximately 10 seconds for CoTI and 2 minutes for the basic QwQ-32B model. <xref ref-type="boxed-text" rid="box1">Textbox 1</xref> shows the overlap between the CoTI- and senior-generated themes. CoTI successfully captured several major themes identified by senior investigators, including adverse drug effects, psychological distress, burden from the number of medications, and burden from the cost of medication.</p>
        <boxed-text id="box1" position="float">
          <title>Comparison of codebook with themes developed by senior investigators and Collaborative Theme Identification Agents (CoTI; without junior investigators) on the heart failure interview transcripts.</title>
          <p>
            <bold>Senior investigator</bold>
          </p>
          <list list-type="bullet">
            <list-item>
              <p>Problems in logistics</p>
            </list-item>
            <list-item>
              <p>Impact from the patient-doctor relations</p>
            </list-item>
          </list>
          <p>
            <bold>Overlap</bold>
          </p>
          <list list-type="bullet">
            <list-item>
              <p>Adverse drug effects</p>
            </list-item>
            <list-item>
              <p>Psychological distress</p>
            </list-item>
            <list-item>
              <p>Burden from the number of medications</p>
            </list-item>
            <list-item>
              <p>Burden from the cost of medications</p>
            </list-item>
          </list>
          <p>
            <bold>CoTI</bold>
          </p>
          <list list-type="bullet">
            <list-item>
              <p>Medication adherence and management challenges</p>
            </list-item>
            <list-item>
              <p>Desire for simplified medication regimen</p>
            </list-item>
            <list-item>
              <p>Perceived effectiveness of medications</p>
            </list-item>
            <list-item>
              <p>Patient-doctor relationship and communication</p>
            </list-item>
          </list>
        </boxed-text>
      </sec>
      <sec>
        <title>CoTI Alone Was More Similar to Senior Investigators Than Junior Investigators or CoTI-Junior Collaboration</title>
        <p>After verifying that CoTI can generate themes more similar to those of senior investigators than other NLP models or the basic LLM, we conducted an exploratory human-AI collaboration experiment to examine whether a human (particularly one with low expertise, such as a junior investigator) could further refine CoTI-generated outputs. The purpose of this experiment was not to compare CoTI with junior investigators performing the full thematic analysis workflow from scratch. Instead, because CoTI is designed to improve the efficiency of thematic analysis by generating clues, initial codes, and themes for human review, we evaluated whether junior investigators could use CoTI-generated outputs as a starting point and refine them.</p>
        <p>We compared three settings (<xref ref-type="table" rid="table4">Table 4</xref>): (1) junior investigator alone, where junior investigators independently identified codes for a subset of assigned transcripts; (2) CoTI alone, where our model extracted clues, identified codes, and generated themes without junior investigators’ feedback for all transcripts; and (3) CoTI+junior investigator, where junior investigators used the CoTI application to review model-generated clues and codes, and provided feedback to CoTI to refine its outputs for each assigned transcript. In all settings, clues, codes, and themes provided by senior investigators served as the reference standard. Since we considered codes as the final analytic output of each transcript, junior investigators were only responsible for manually identifying codes and did not perform clue extraction. Additionally, because the themes can be generated only after considering codes across all transcripts and junior investigators were assigned only a subset of transcripts, they could not construct a complete codebook with themes. Therefore, the evaluation focused on clue extraction for the CoTI alone and CoTI+junior investigator settings, and on code identification for all 3 settings.</p>
        <table-wrap position="float" id="table4">
          <label>Table 4</label>
          <caption>
            <p>Four different settings to perform thematic analysis on the heart failure interview transcripts.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="250"/>
            <col width="300"/>
            <col width="150"/>
            <col width="150"/>
            <col width="150"/>
            <thead>
              <tr valign="top">
                <td>Setting</td>
                <td>Interview transcripts to process (N=12), n</td>
                <td>Extract clues</td>
                <td>Identify codes</td>
                <td>Generate themes</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Junior investigator alone</td>
                <td>7</td>
                <td>No</td>
                <td>Yes</td>
                <td>No</td>
              </tr>
              <tr valign="top">
                <td>CoTI<sup>a</sup> alone</td>
                <td>12</td>
                <td>Yes</td>
                <td>Yes</td>
                <td>Yes</td>
              </tr>
              <tr valign="top">
                <td>CoTI+junior investigator</td>
                <td>7</td>
                <td>Yes</td>
                <td>Yes</td>
                <td>No</td>
              </tr>
              <tr valign="top">
                <td>Senior investigators only (reference standard)</td>
                <td>12</td>
                <td>Yes</td>
                <td>Yes</td>
                <td>Yes</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table4fn1">
              <p><sup>a</sup>CoTI: Collaborative Theme Identification Agents.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <p>To evaluate whether junior investigators’ feedback could help CoTI to extract clues that are more similar to those of the senior investigators, we compared clue similarity between CoTI alone and the senior investigators, and between CoTI with junior investigators’ feedback and the senior investigators. Among the evaluation metrics, we prioritized recall as the most meaningful evaluation metric since higher recall indicates that the model successfully retrieves a large proportion of relevant clues that were identified by senior investigators, which is critical for ensuring that subsequent code identification is grounded in a sufficiently rich evidence base. Although precision reflects the accuracy of extracted clues, occasional inclusion of clues not identified by senior investigators may be less damaging because qualitative thematic analysis is inherently subjective; additional identified results may also be valid even if they are not present in the reference standard. As shown in <xref rid="figure4" ref-type="fig">Figure 4</xref>A, CoTI with junior investigators’ feedback resulted in small or modest recall improvements in many transcripts. While precision was not the primary metric, it offers complementary insight into the correctness of extracted clues. As shown in <xref rid="figure4" ref-type="fig">Figure 4</xref>B, CoTI with junior investigators’ feedback generally yielded lower precision compared to CoTI alone. Even in the few cases where precision improved after junior investigators’ feedback, such as transcripts 3 and 7, the gains were marginal. These results suggest that CoTI alone already provides strong clue similarity, and junior investigators’ feedback brings limited or even negative impact on improving clue similarity with senior investigators. One possible hypothesis is that the gap between CoTI and senior investigators may involve domain-specific understanding that junior investigators also lack. As a result, junior investigators’ feedback may not sufficiently refine the model’s outputs. Another possible hypothesis is that junior investigators may have exhibited automation bias, a tendency to rely on outputs from CoTI, and this overreliance could have reduced their critical engagement with the model’s outputs.</p>
        <p>We continued evaluating the impact of junior investigators’ feedback for the task of code identification. As shown in <xref rid="figure4" ref-type="fig">Figure 4</xref>C, codes identified by both CoTI alone and CoTI with junior investigators’ feedback achieved higher similarity to senior investigators than codes manually identified by junior investigators in many cases. Notably, for some cases, incorporating junior investigators’ feedback to CoTI led to identified codes that were more similar to those of senior investigators, as seen in transcripts 2, 3, 4, and 8, where CoTI alone achieved moderate theme similarity scores (approximately between 0.35 and 0.40). However, when CoTI alone already achieved strong code similarity to senior investigators (cosine similarity ≥ 0.45), incorporating junior investigators’ feedback tended to reduce that similarity, as observed in transcripts 5, 6, and 10. Compared to the clue extraction task, junior investigators’ feedback appeared more helpful in supporting the code identification task, suggesting junior investigators may be more adept at higher-level interpretation than at intermediate clue extraction.</p>
        <p>The marginal benefit of junior investigators’ feedback to CoTI might be due to junior investigators’ perception of AI. To explore this possibility, we collected their perceptions of CoTI. As illustrated in <xref rid="figure5" ref-type="fig">Figure 5</xref>A, subjective ratings were generally high across all junior investigators, indicating strong approval of CoTI’s accuracy, relevance, trustworthiness, and overall satisfaction. These favorable perceptions suggest that junior investigators may have felt less need to critically revise or question CoTI’s outputs, thereby reducing the additive value of AI-human collaboration. However, we noticed several exceptions. Investigator D rated CoTI as missing codes in every assigned transcript, resulting in a code comprehensiveness score of 0%. This is particularly striking given that D simultaneously gave perfect trust (100%) and top scores across most other dimensions. To better understand this contradiction, we analyzed D’s performance in the manual code identification task. D achieved a code similarity of 0.41, higher than investigators A and C, though slightly lower than CoTI alone (code similarity=0.45). Additionally, D identified a total of 47 codes across assigned interviews, compared to 41 themes identified by CoTI. These results suggested that investigator D was particularly context-sensitive and may have maintained a high bar for code completeness. Although D trusted CoTI’s identified codes, D likely expected CoTI not only to identify obvious codes but also to capture more implicit ones. In addition, collaboration between D and CoTI led to small gains in clue similarity compared to CoTI alone, which indicated that even for a higher standard evaluator, collaboration with AI offered marginal but observable benefits. Another notable exception was investigator A, whose performance in code identification was comparatively weaker than investigators B and D. Moreover, CoTI alone achieved higher code similarity than A’s individual performance, and collaboration between A and CoTI improved clue similarity compared to CoTI alone. Nevertheless, investigator A reported lower trust in CoTI compared with other investigators, indicating that A’s confidence in AI remained low even when CoTI outperformed A’s own performance and the collaboration yielded measurable benefits.</p>
        <fig id="figure4" position="float">
          <label>Figure 4</label>
          <caption>
            <p>Evaluation of clue extraction and code identification across interviews. (A) Recall of extracted clues for Collaborative Theme Identification Agents (CoTI)–only and CoTI+human settings, evaluated against clues identified by senior investigators. (B) Precision of extracted clues for CoTI-only and CoTI+human settings, evaluated against clues identified by senior investigators. (C) Cosine similarity between codes generated by human-only, CoTI-only, or CoTI+human, compared to senior investigators’ identified codes (reference standard).</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e90872_fig4.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <fig id="figure5" position="float">
          <label>Figure 5</label>
          <caption>
            <p>Subjective survey and objective evaluations by junior investigators. (A) This spider chart displays subjective perceptions from 4 junior investigators, A-D, regarding Collaborative Theme Identification Agents (CoTI). Subjective measures include ratings of clue accuracy and relevance, reasoning clarity and logic, code accuracy, relevance, and comprehensiveness, as well as trust and overall satisfaction. (B) This spider chart displays objective performance by junior investigators and CoTI, including clue recall (CoTI alone), clue recall (CoTI+junior investigator), code cosine similarity (junior investigator alone), code cosine similarity (CoTI alone), and code cosine similarity (CoTI+junior investigator).</p>
          </caption>
          <graphic xlink:href="jmir_v28i1e90872_fig5.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Ablation Study Highlights Components’ Contributions to CoTI Performance</title>
        <p>Aiming to understand the contributions of each component within our model, we conducted an ablation study (<xref ref-type="table" rid="table5">Table 5</xref>). The study began with a GPT-4o-mini model, which served as the baseline. In this setting, GPT-4o-mini performed thematic analysis only one time for each transcript (total 12). Introducing Instructor, an agent designed to iteratively refine instruction prompts, improved the similarity of extracted clues (Jaccard similarity: +14.3%; precision: +24.3%; recall: +2.7%; <italic>F</italic><sub>1</sub>-score: +11.0%). These results highlighted the critical role of high-quality, tailored instruction prompts in improving CoTI’s ability to extract clues that were more similar to those extracted by senior investigators. Subsequently, we applied a multirun aggregation strategy, which repeated the code identification, including clue extraction, 3 times and retained the union of the resulting outputs. This strategy further improved performance, yielding relatively greater gains in clue similarity (Jaccard similarity: +17.5%; recall: +18.7%; <italic>F</italic><sub>1</sub>-score: +12.5%) with modest improvements in code similarity (cosine similarity: +0.9%). These findings indicated that multiple code identification passes allowed our model to capture a broader range of codes and supporting clues, including those that may be missed in a single run due to the randomness in LLM outputs.</p>
        <table-wrap position="float" id="table5">
          <label>Table 5</label>
          <caption>
            <p>Overview of Collaborative Theme Identification Agents (CoTI) components and additive contributions to the performance.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="150"/>
            <col width="140"/>
            <col width="140"/>
            <col width="140"/>
            <col width="140"/>
            <col width="0"/>
            <col width="140"/>
            <col width="0"/>
            <col width="150"/>
            <thead>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td colspan="5">Clue extraction</td>
                <td colspan="2">Code identification</td>
                <td>Response time for each transcript</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Jaccard similarity (relative change, %)<sup>a</sup></td>
                <td>Precision (relative change, %)<sup>a</sup></td>
                <td>Recall (relative change, %)<sup>a</sup></td>
                <td><italic>F</italic><sub>1</sub>-score (relative change, %)<sup>a</sup></td>
                <td colspan="2">Cosine similarity (relative change, %)<sup>a</sup></td>
                <td colspan="2">Time (seconds)</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>GPT-4o-mini</td>
                <td>0.300 (N/A<sup>b</sup>)</td>
                <td>0.441 (N/A)</td>
                <td>0.517 (N/A)</td>
                <td>0.456 (N/A)</td>
                <td colspan="2">0.433 (N/A)</td>
                <td colspan="2">~8</td>
              </tr>
              <tr valign="top">
                <td>GPT-4o-mini with Instructor</td>
                <td>0.343 (+14.33)</td>
                <td>0.548 (+24.26)</td>
                <td>0.531 (+2.71)</td>
                <td>0.506 (+10.96)</td>
                <td colspan="2">0.427 (–1.39)</td>
                <td colspan="2">~20</td>
              </tr>
              <tr valign="top">
                <td>CoTI (GPT-4o-mini with Instructor and multirun aggregation)</td>
                <td>0.403 (+17.49)</td>
                <td>0.550 (+0.36)</td>
                <td>0.630 (+18.64)</td>
                <td>0.569 (+12.45)</td>
                <td colspan="2">0.431 (+0.94)</td>
                <td colspan="2">~60</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table5fn1">
              <p><sup>a</sup>Values indicate the relative change (%) compared with the preceding model configuration.</p>
            </fn>
            <fn id="table5fn2">
              <p><sup>b</sup>N/A: not applicable.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>Generalizability of CoTI on External Datasets</title>
        <p>To further evaluate the generalizability of CoTI, we applied our framework to several different health care qualitative datasets, including the health system’s response to COVID-19 in Sierra Leone, which comprised 21 interview transcripts [<xref ref-type="bibr" rid="ref25">25</xref>], and the family carers’ strategies when a family member with dementia was agitated, which comprised 18 interview transcripts [<xref ref-type="bibr" rid="ref29">29</xref>]. Table S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> shows the computational evaluations among these external datasets. Tables S5 and S6 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> show overlaps between the themes generated by our model and by experts. Across these external datasets, CoTI achieved higher clue similarity than theme similarity, indicating that it identified relevant supporting excerpts more effectively than it reproduced expert-level themes. This suggests that human researcher review remains necessary to interpret and validate generated themes.</p>
      </sec>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <p>Our study demonstrates the efficiency of leveraging multiagent LLMs to support thematic analysis in qualitative research. By integrating prompt engineering with multiagent collaboration, our proposed framework successfully extracted supporting excerpts, identified initial codes, and developed a codebook with themes through an automated process. Beyond its application in heart failure qualitative research, our framework has demonstrated its efficiency in other domains, such as COVID-19–related and Alzheimer-related studies. In addition, our developed CoTI interactive application allows human researchers to rapidly review model-generated outputs, locate key transcript sections by searching keywords derived from model-extracted supporting excerpts, and provide feedback to refine the outputs. This interactive design facilitates human-AI collaboration and may help researchers more efficiently gain insights into patients’ experiences, thereby facilitating qualitative research. However, our preliminary experiment found that collaboration between CoTI and junior investigators yielded marginal improvements over CoTI alone in the heart failure study.</p>
      <p>The computational evaluations showed that the themes generated by CoTI alone were more similar to those identified by senior investigators than those generated by junior investigators working independently or by baseline models. To further examine the quality of these themes, we incorporated an expert qualitative appraisal as a complementary assessment. The expert identified overlap among the themes “medication burden,” “psychological and emotional impact of medications,” and “side effects and their perception.” In particular, the expert considered “side effects and their perception” unnecessary as a separate theme because these experiences can be integrated within either medication burden or the psychological and emotional impact of medications. Similarly, the expert noted that “desire for simplified medication regimen” can be integrated into “medication adherence and management challenges.” These observations suggested that CoTI may separate related experiences into multiple themes, resulting in thematic distinctiveness that is less clearly differentiated. This qualitative appraisal aligns with our computational evaluation, which indicated relatively modest theme distinctiveness (0.473).</p>
      <p>Next, we compared CoTI-generated themes with the reference themes developed by senior investigators to identify where CoTI missed or partially distorted the reference interpretations. One missed example was the reference theme “problems in logistics.” In senior investigators' generated interpretations, this theme referred not only to medication-taking behavior but also to broader practical barriers, including transportation difficulties, obtaining medications, coordinating refills, identifying the correct pharmacy, and relying on family members or caregivers to complete multiple steps in the medication process. CoTI did not generate an equivalent theme. Instead, it generated the theme “medication adherence and management challenges,” which focused more narrowly on patients’ difficulties with medication adherence, including forgetfulness, confusion, overdosing, and the need for structured support systems. One distorted example involved the reference theme “impact from patient-doctor relations.” The senior investigators' generated theme captured a complex pattern that included both trust and dependency on physicians, as well as skepticism toward pharmaceutical companies and insurance systems. CoTI generated a related theme, “patient-doctor relationship and communication,” but this theme emphasized communication gaps and patient education rather than the broader trust-mistrust tension captured in the reference interpretations. Another partially distorted example was CoTI’s theme “desire for simplified medication regimen.” Although the supporting excerpts for this theme overlapped with those extracted by senior investigators, the senior investigators interpreted these excerpts as part of the broader theme “burden from the number of medications.” In contrast, CoTI considered patients’ preferences for a simplified or reduced medication regimen as a separate theme, rather than integrating them into medication burden. CoTI also generated a theme “perceived effectiveness of medications,” which was plausible but not supported by the reference interpretations. Although some patients discussed whether their medications were working, the senior investigators’ generated themes did not identify it as a separate theme. This suggested that CoTI may sometimes identify some less important patterns as distinct themes even when human researchers interpret them as secondary or insufficiently central to the overall research question. Together, these examples suggest that while CoTI can identify relevant semantic content, it may sometimes reorganize these patterns into themes that are more general, more familiar, or less central to the research question than those developed by human researchers. These findings indicate that CoTI can serve as an efficient tool to support thematic analysis, but its outputs still require careful human review and interpretation. This conclusion is consistent with Shanwetter et al [<xref ref-type="bibr" rid="ref15">15</xref>], who emphasized that LLMs should complement rather than replace human analysis, particularly when identifying structural themes.</p>
      <p>Our study also provides preliminary insights into the potential role of CoTI in supporting human-AI collaboration in thematic analysis. First, CoTI alone produced codes more similar to senior investigators than junior investigators. This highlights CoTI’s potential as a reliable tool for thematic analysis, particularly when senior investigators are unavailable. Second, collaboration with junior investigators did not consistently improve clue or code similarity to senior investigators. While human-AI collaboration is often assumed to enhance AI outputs, our results show that the effectiveness is marginal. Specifically, when CoTI alone code similarity was moderate to senior investigators, junior investigators’ feedback could help refine codes and enhance similarity. However, when CoTI alone already performed strongly, such feedback often degraded code similarity. In addition, CoTI alone extracted clues that often achieved higher recall and precision; however, incorporating junior investigators’ feedback sometimes made clue similarity decrease. Survey analyses indicated that the marginal benefits of collaboration may stem from junior investigators’ overreliance on CoTI’s outputs, which could cause automation bias and reduce their independent critical thinking.</p>
      <p>Our study has several limitations. First, the human-AI collaboration was exploratory. We recruited only 4 junior investigators, who analyzed selected subsets of transcripts and did not complete the full thematic analysis workflow, including excerpt extraction, code identification, and theme generation through a consensus or adjudication process. Therefore, our findings only provided preliminary evidence that CoTI-generated outputs were more similar to the senior investigators’ generated reference standard than those generated by the participating junior investigators. Future studies should evaluate both CoTI alone and CoTI-assisted human-AI collaboration against rigorous human-led thematic analysis workflows involving multiple trained coders, consensus coding, and adjudication, thereby providing a more rigorous assessment of CoTI’s effectiveness and the potential benefits of human-AI collaboration in thematic analysis. Second, the human-AI collaboration was conducted on a single case study of heart failure interviews, which limits the generalizability of our findings on human-AI collaboration in thematic analysis. Further validation across other research contexts is needed to establish broader generalizability. Moreover, the potential impact of incorporating feedback from senior investigators into CoTI was not evaluated, which may limit understanding of how senior investigators’ input could further enhance our model performance. Another limitation is that our interpretation of automation bias was based on a small exploratory survey of junior investigators’ perceptions of CoTI. Although the survey results suggested that the high trust in CoTI may reduce junior investigators’ tendency to critically revise AI-generated outputs, these findings remain preliminary because of the small sample size and reliance on self-reported perceptions. Moreover, alternative explanations cannot be excluded, such as limited qualitative analysis experience among junior investigators and lower quality of junior investigators’ feedback. Therefore, we cannot consider automation bias or other factors as the definitive explanations for the limited additive benefit of junior investigators’ involvement.</p>
      <p>In all, CoTI provides preliminary evidence that multiagent LLMs can replicate key phases of qualitative thematic analysis while maintaining human involvement through review and refinement. In addition, our study suggests that the value of LLMs in qualitative research lies not in replacing researchers, but in expanding their capacity to analyze larger and more complex datasets efficiently while maintaining interpretive oversight. Future research should focus not only on improving model performance, but also on designing collaborative workflows that preserve reflexivity and critical interpretation.</p>
    </sec>
  </body>
  <back>
    <app-group>
      <supplementary-material id="app1">
        <label>Multimedia Appendix 1</label>
        <p>Supplementary tables and detailed prompt design of the large language model workflow.</p>
        <media xlink:href="jmir_v28i1e90872_app1.docx" xlink:title="DOCX File , 39 KB"/>
      </supplementary-material>
    </app-group>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">CoTI</term>
          <def>
            <p>Collaborative Theme Identification Agents</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">IRB</term>
          <def>
            <p>institutional review board</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb3">LDA</term>
          <def>
            <p>latent Dirichlet allocation</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb4">LLM</term>
          <def>
            <p>large language model</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb5">NLP</term>
          <def>
            <p>natural language processing</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb6">QUEST</term>
          <def>
            <p>quality, understanding, expression, safety, and trust</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>The authors used generative artificial intelligence (ChatGPT) to assist with manuscript language refinement. All AI-assisted text was reviewed, edited, and verified by the authors, who take responsibility for the accuracy and integrity.</p>
    </ack>
    <notes>
      <sec>
        <title>Funding</title>
        <p>This work was supported in part by the National Institutes of Health (NIH) under award numbers R01AG082721 and R01AG084637.</p>
      </sec>
    </notes>
    <notes>
      <sec>
        <title>Data Availability</title>
        <p>The interview transcripts about patients with heart failure cannot be made publicly available due to privacy concerns. For interview transcripts about COVID-19, please see Open Science Framework [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref30">30</xref>]. For interview transcripts about Alzheimer-related topics, please see <ext-link ext-link-type="uri" xlink:href="https://data.mendeley.com/datasets/s8wtptnhyc/1%20%5b28" xlink:type="simple">Mendeley Data [28</ext-link>,30]. The code used in this study is available on GitHub [<xref ref-type="bibr" rid="ref32">32</xref>].</p>
      </sec>
    </notes>
    <fn-group>
      <fn fn-type="con">
        <p>Conceptualization: QX, YK</p>
        <p>Methodology: QX, YK</p>
        <p>Investigation: NA, MJK, GG, AC, DH, AW</p>
        <p>Formal analysis: QX, YK</p>
        <p>Supervision: YK</p>
        <p>Writing—original draft: QX, MJK, YK</p>
        <p>Writing—review and editing: all authors</p>
      </fn>
      <fn fn-type="conflict">
        <p>MJK received a consulting fee from Novo Nordisk. Otherwise, no potential conflict of interest relevant to this article was reported.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Barry</surname>
              <given-names>MJ</given-names>
            </name>
            <name name-style="western">
              <surname>Edgman-Levitan</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Shared decision making--pinnacle of patient-centered care</article-title>
          <source>N Engl J Med</source>
          <year>2012</year>
          <volume>366</volume>
          <issue>9</issue>
          <fpage>780</fpage>
          <lpage>781</lpage>
          <pub-id pub-id-type="doi">10.1056/NEJMp1109283</pub-id>
          <pub-id pub-id-type="medline">22375967</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hudon</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Fortin</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Haggerty</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Loignon</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Lambert</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Poitras</surname>
              <given-names>ME</given-names>
            </name>
          </person-group>
          <article-title>Patient-centered care in chronic disease management: a thematic analysis of the literature in family medicine</article-title>
          <source>Patient Educ Couns</source>
          <year>2012</year>
          <volume>88</volume>
          <issue>2</issue>
          <fpage>170</fpage>
          <lpage>176</lpage>
          <pub-id pub-id-type="doi">10.1016/j.pec.2012.01.009</pub-id>
          <pub-id pub-id-type="medline">22360841</pub-id>
          <pub-id pub-id-type="pii">S0738-3991(12)00040-7</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Grudniewicz</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Gray</surname>
              <given-names>CS</given-names>
            </name>
            <name name-style="western">
              <surname>Boeckxstaens</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>De Maeseneer</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Mold</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Operationalizing the chronic care model with goal-oriented care</article-title>
          <source>Patient</source>
          <year>2023</year>
          <volume>16</volume>
          <issue>6</issue>
          <fpage>569</fpage>
          <lpage>578</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/37642918"/>
          </comment>
          <pub-id pub-id-type="doi">10.1007/s40271-023-00645-8</pub-id>
          <pub-id pub-id-type="medline">37642918</pub-id>
          <pub-id pub-id-type="pii">10.1007/s40271-023-00645-8</pub-id>
          <pub-id pub-id-type="pmcid">PMC10570240</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gill</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Stewart</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Treasure</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Chadwick</surname>
              <given-names>B</given-names>
            </name>
          </person-group>
          <article-title>Methods of data collection in qualitative research: interviews and focus groups</article-title>
          <source>Br Dent J</source>
          <year>2008</year>
          <volume>204</volume>
          <issue>6</issue>
          <fpage>291</fpage>
          <lpage>295</lpage>
          <pub-id pub-id-type="doi">10.1038/bdj.2008.192</pub-id>
          <pub-id pub-id-type="medline">18356873</pub-id>
          <pub-id pub-id-type="pii">bdj.2008.192</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Vaismoradi</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Jones</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Turunen</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Snelgrove</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Theme development in qualitative content analysis and thematic analysis</article-title>
          <source>J Nurs Educ Pract</source>
          <year>2016</year>
          <volume>6</volume>
          <issue>5</issue>
          <pub-id pub-id-type="doi">10.5430/jnep.v6n5p100</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Amjad</surname>
              <given-names>NA</given-names>
            </name>
            <name name-style="western">
              <surname>Shoar</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Bryant</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Hunt</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Kwak</surname>
              <given-names>MJ</given-names>
            </name>
          </person-group>
          <article-title>Perceptions of the intensity of heart failure medications among hospitalized older adults: a pilot qualitative study</article-title>
          <source>Ann Geriatr Med Res</source>
          <year>2025</year>
          <volume>29</volume>
          <issue>2</issue>
          <fpage>233</fpage>
          <lpage>239</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://e-agmr.org/journal/view.php?doi=10.4235/agmr.24.0182"/>
          </comment>
          <pub-id pub-id-type="doi">10.4235/agmr.24.0182</pub-id>
          <pub-id pub-id-type="medline">39945131</pub-id>
          <pub-id pub-id-type="pii">agmr.24.0182</pub-id>
          <pub-id pub-id-type="pmcid">PMC12215014</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Nordfonn</surname>
              <given-names>OK</given-names>
            </name>
            <name name-style="western">
              <surname>Morken</surname>
              <given-names>IM</given-names>
            </name>
            <name name-style="western">
              <surname>Lunde Husebø</surname>
              <given-names>AM</given-names>
            </name>
          </person-group>
          <article-title>A qualitative study of living with the burden from heart failure treatment: exploring the patient capacity for self-care</article-title>
          <source>Nurs Open</source>
          <year>2020</year>
          <volume>7</volume>
          <issue>3</issue>
          <fpage>804</fpage>
          <lpage>813</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/32257268"/>
          </comment>
          <pub-id pub-id-type="doi">10.1002/nop2.455</pub-id>
          <pub-id pub-id-type="medline">32257268</pub-id>
          <pub-id pub-id-type="pii">NOP2455</pub-id>
          <pub-id pub-id-type="pmcid">PMC7113501</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Leeson</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Resnick</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Alexander</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Rovers</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Natural language processing (NLP) in qualitative public health research: a proof of concept study</article-title>
          <source>Int J Qual Methods</source>
          <year>2019</year>
          <volume>18</volume>
          <fpage>160940691988702</fpage>
          <pub-id pub-id-type="doi">10.1177/1609406919887021</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Blei</surname>
              <given-names>DM</given-names>
            </name>
          </person-group>
          <article-title>Latent Dirichlet allocation</article-title>
          <source>J Mach Learn Res</source>
          <year>2003</year>
          <volume>3</volume>
          <fpage>993</fpage>
          <lpage>1022</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmlr.org/papers/volume3/blei03a/blei03a.pdf"/>
          </comment>
          <pub-id pub-id-type="doi">10.7551/mitpress/1120.003.0082</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Abram</surname>
              <given-names>MD</given-names>
            </name>
            <name name-style="western">
              <surname>Mancini</surname>
              <given-names>KT</given-names>
            </name>
            <name name-style="western">
              <surname>Parker</surname>
              <given-names>RD</given-names>
            </name>
          </person-group>
          <article-title>Methods to integrate natural language processing into qualitative research</article-title>
          <source>Int J Qual Methods</source>
          <year>2020</year>
          <volume>19</volume>
          <fpage>160940692098460</fpage>
          <pub-id pub-id-type="doi">10.1177/1609406920984608</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="web">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Grootendorst</surname>
              <given-names>MP</given-names>
            </name>
          </person-group>
          <article-title>Supervised topic modeling</article-title>
          <source>GitHub</source>
          <access-date>2026-09-10</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://maartengr.github.io/BERTopic/getting_started/supervised/supervised.html">https://maartengr.github.io/BERTopic/getting_started/supervised/supervised.html</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <collab>OpenAI</collab>
            <name name-style="western">
              <surname>Achiam</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Adler</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Agarwal</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Ahmad</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Akkaya</surname>
              <given-names>I</given-names>
            </name>
          </person-group>
          <article-title>GPT-4 technical report</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on March 15, 2023</comment>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://arxiv.org/abs/2303.08774"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Renard</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>LaNoue</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Østbye</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Boehm</surname>
              <given-names>LM</given-names>
            </name>
            <name name-style="western">
              <surname>Wahid</surname>
              <given-names>L</given-names>
            </name>
          </person-group>
          <article-title>Mary steps out: capturing patient experience through qualitative and AI methods</article-title>
          <source>NEJM AI</source>
          <year>2024</year>
          <volume>1</volume>
          <issue>12</issue>
          <fpage>1</fpage>
          <pub-id pub-id-type="doi">10.1056/aip2400567</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Mannstadt</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Goodman</surname>
              <given-names>SM</given-names>
            </name>
            <name name-style="western">
              <surname>Rajan</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Young</surname>
              <given-names>SR</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Navarro-Millán</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Mehta</surname>
              <given-names>B</given-names>
            </name>
          </person-group>
          <article-title>A novel approach for mixed-methods research using large language models: a report using patients' perspectives on barriers to arthroplasty</article-title>
          <source>ACR Open Rheumatol</source>
          <year>2024</year>
          <volume>6</volume>
          <issue>6</issue>
          <fpage>375</fpage>
          <lpage>379</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/38454175"/>
          </comment>
          <pub-id pub-id-type="doi">10.1002/acr2.11662</pub-id>
          <pub-id pub-id-type="medline">38454175</pub-id>
          <pub-id pub-id-type="pmcid">PMC11168905</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Shanwetter Levit</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Saban</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>When investigator meets large language models: a qualitative analysis of cancer patient decision-making journeys</article-title>
          <source>NPJ Digit Med</source>
          <year>2025</year>
          <volume>8</volume>
          <issue>1</issue>
          <fpage>336</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41746-025-01747-3"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-025-01747-3</pub-id>
          <pub-id pub-id-type="medline">40473767</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-025-01747-3</pub-id>
          <pub-id pub-id-type="pmcid">PMC12141523</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Qiao</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Walker</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Cunningham</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Koh</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>Thematic-LM: a LLM-based multi-agent system for large-scale thematic analysis</article-title>
          <year>2025</year>
          <conf-name>WWW '25: Proceedings of the ACM on Web Conference 2025</conf-name>
          <conf-date>2025 April 28</conf-date>
          <conf-loc>Sydney NSW Australia</conf-loc>
          <publisher-name>ACM</publisher-name>
          <fpage>649</fpage>
          <lpage>658</lpage>
          <pub-id pub-id-type="doi">10.1145/3696410.3714595</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Yi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Lim</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Well</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Mery</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ji</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Pingali</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Leng</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Ding</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>TAMA: a human-AI collaborative thematic analysis framework using multi-agent LLMs for clinical interviews</article-title>
          <source>ACM Trans Comput Healthcare</source>
          <year>2026</year>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://arxiv.org/abs/2503.20666"/>
          </comment>
          <pub-id pub-id-type="doi">10.1145/3828752</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref18">
        <label>18</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yi</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Nguyen</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Lim</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Well</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Markey</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Auto-TA: towards scalable automated thematic analysis (TA) via multi-agent large language models with reinforcement learning</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on June 30, 2025</comment>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://arxiv.org/abs/2506.23998"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref19">
        <label>19</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Radford</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Brockman</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>McLeavey</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Sutskever</surname>
              <given-names>I</given-names>
            </name>
          </person-group>
          <article-title>Robust speech recognition via large-scale weak supervision</article-title>
          <source>ICML</source>
          <year>2022</year>
          <fpage>28492</fpage>
          <lpage>28518</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://cdn.openai.com/papers/whisper.pdf"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref20">
        <label>20</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Braun</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Clarke</surname>
              <given-names>V</given-names>
            </name>
          </person-group>
          <article-title>Using thematic analysis in psychology</article-title>
          <source>Qual Res Psychol</source>
          <year>2008</year>
          <volume>3</volume>
          <issue>2</issue>
          <fpage>77</fpage>
          <lpage>101</lpage>
          <pub-id pub-id-type="doi">10.1191/1478088706qp063oa</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref21">
        <label>21</label>
        <nlm-citation citation-type="web">
          <article-title>QwQ-32B: embracing the power of reinforcement learning</article-title>
          <source>GitHub</source>
          <year>2025</year>
          <access-date>2026-09-10</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://qwenlm.github.io/blog/qwq-32b/">https://qwenlm.github.io/blog/qwq-32b/</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref22">
        <label>22</label>
        <nlm-citation citation-type="web">
          <article-title>GPT-4o mini: advancing cost-efficient intelligence</article-title>
          <source>OpenAI</source>
          <access-date>2026-09-10</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://openai.com/index/gpt-4o-mini-advancing-cost-efficient-intelligence/">https://openai.com/index/gpt-4o-mini-advancing-cost-efficient-intelligence/</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref23">
        <label>23</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Guo</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>Text classification via large language models</article-title>
          <source>arXiv [cs.CL]</source>
          <comment>Preprint posted online on October 9, 2023</comment>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://arxiv.org/abs/2305.08377"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref24">
        <label>24</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yuksekgonul</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Bianchi</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Boen</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Guestrin</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>TextGrad: automatic "differentiation" via text</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on June 11, 2024</comment>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://arxiv.org/abs/2406.07496"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref25">
        <label>25</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Stone</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Bailey</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Wurie</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Leather</surname>
              <given-names>AJM</given-names>
            </name>
            <name name-style="western">
              <surname>Davies</surname>
              <given-names>JI</given-names>
            </name>
            <name name-style="western">
              <surname>Bolkan</surname>
              <given-names>HA</given-names>
            </name>
            <name name-style="western">
              <surname>Sevalie</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Youkee</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Parmar</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>A qualitative study examining the health system's response to COVID-19 in Sierra Leone</article-title>
          <source>PLoS One</source>
          <year>2024</year>
          <volume>19</volume>
          <issue>2</issue>
          <fpage>e0294391</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://dx.plos.org/10.1371/journal.pone.0294391"/>
          </comment>
          <pub-id pub-id-type="doi">10.1371/journal.pone.0294391</pub-id>
          <pub-id pub-id-type="medline">38306321</pub-id>
          <pub-id pub-id-type="pii">PONE-D-22-20556</pub-id>
          <pub-id pub-id-type="pmcid">PMC10836672</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref26">
        <label>26</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Tam</surname>
              <given-names>TYC</given-names>
            </name>
            <name name-style="western">
              <surname>Sivarajkumar</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kapoor</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Stolyar</surname>
              <given-names>AV</given-names>
            </name>
            <name name-style="western">
              <surname>Polanska</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>McCarthy</surname>
              <given-names>KR</given-names>
            </name>
            <name name-style="western">
              <surname>Osterhoudt</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Visweswaran</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Fu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Mathur</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Cacciamani</surname>
              <given-names>GE</given-names>
            </name>
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Peng</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>A framework for human evaluation of large language models in healthcare derived from literature review</article-title>
          <source>NPJ Digit Med</source>
          <year>2024</year>
          <volume>7</volume>
          <issue>1</issue>
          <fpage>258</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41746-024-01258-7"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-024-01258-7</pub-id>
          <pub-id pub-id-type="medline">39333376</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-024-01258-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC11437138</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref27">
        <label>27</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Angelov</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Top2Vec: distributed representations of topics</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on August 19, 2020</comment>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://arxiv.org/abs/2008.09470"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref28">
        <label>28</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Grootendorst</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>BERTopic: neural topic modeling with a class-based TF-IDF procedure</article-title>
          <source>arXiv</source>
          <comment>Preprint posted online on March 11, 2022</comment>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://arxiv.org/abs/2203.05794"/>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref29">
        <label>29</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hoe</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Jesnick</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Turner</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Leavey</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Livingston</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>Caring for relatives with agitation at home: a qualitative study of positive coping strategies</article-title>
          <source>BJPsych Open</source>
          <year>2017</year>
          <volume>3</volume>
          <issue>1</issue>
          <fpage>34</fpage>
          <lpage>40</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/28243464"/>
          </comment>
          <pub-id pub-id-type="doi">10.1192/bjpo.bp.116.004069</pub-id>
          <pub-id pub-id-type="medline">28243464</pub-id>
          <pub-id pub-id-type="pii">S2056472400002003</pub-id>
          <pub-id pub-id-type="pmcid">PMC5299384</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref30">
        <label>30</label>
        <nlm-citation citation-type="web">
          <article-title>A qualitative study examining the health system’s response to COVID-19 in Sierra Leone</article-title>
          <source>OSF</source>
          <access-date>2026-09-16</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://osf.io/qc3z8/overview">https://osf.io/qc3z8/overview</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref31">
        <label>31</label>
        <nlm-citation citation-type="web">
          <article-title>Managing agitation and raising quality of life: semi structured interviews with family carers of people living with dementia</article-title>
          <source>Mendeley Data</source>
          <access-date>2026-09-16</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://data.mendeley.com/datasets/s8wtptnhyc/1">https://data.mendeley.com/datasets/s8wtptnhyc/1</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref32">
        <label>32</label>
        <nlm-citation citation-type="web">
          <article-title>CoTI-MultiAgent-Theme-Analysis</article-title>
          <source>GitHub</source>
          <access-date>2026-09-15</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://github.com/QidiXu96/CoTI-MultiAgent-Theme-Analysis">https://github.com/QidiXu96/CoTI-MultiAgent-Theme-Analysis</ext-link>
          </comment>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
