<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e98519</article-id><article-id pub-id-type="doi">10.2196/98519</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Retrieval-Augmented Large Language Model Counseling for Continuous Glucose Monitoring in Diabetes: Source-Masked Multirater Comparative Evaluation</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Guo</surname><given-names>Zhijun</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Lai</surname><given-names>Alvina</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Korakas</surname><given-names>Emmanouil</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Vagenas</surname><given-names>Aristeidis</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ahamed</surname><given-names>Irshad</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Albor</surname><given-names>Christo</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Hengrui</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Healy</surname><given-names>Justin</given-names></name><degrees>MBBS, MPH</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Li</surname><given-names>Kezhi</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib></contrib-group><aff id="aff1"><institution>Institute of Health Informatics, University College London</institution><addr-line>222 Euston Road</addr-line><addr-line>London</addr-line><country>United Kingdom</country></aff><aff id="aff2"><institution>University College London Hospitals NHS Foundation Trust, London, United Kingdom</institution><addr-line>London</addr-line><country>United Kingdom</country></aff><aff id="aff3"><institution>Dartford and Gravesham NHS Foundation Trust, Dartford, United Kingdom</institution><addr-line>Dartford</addr-line><addr-line>England</addr-line><country>United Kingdom</country></aff><aff id="aff4"><institution>Royal Free London NHS Foundation Trust</institution><addr-line>London</addr-line><country>United Kingdom</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Facchinetti</surname><given-names>Andrea</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Liu</surname><given-names>Zhao</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Kezhi Li, PhD, Institute of Health Informatics, University College London, 222 Euston Road, London, NW1 2DA, United Kingdom, 44 7859995590; <email>ken.li@ucl.ac.uk</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>31</day><month>7</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e98519</elocation-id><history><date date-type="received"><day>16</day><month>04</month><year>2026</year></date><date date-type="rev-recd"><day>30</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>30</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Zhijun Guo, Alvina Lai, Emmanouil Korakas, Aristeidis Vagenas, Irshad Ahamed, Christo Albor, Hengrui Zhang, Justin Healy, Kezhi Li. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 31.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e98519"/><abstract><sec><title>Background</title><p>Continuous glucose monitoring (CGM) is central to diabetes care, but explaining CGM patterns consistently and empathetically remains time-intensive in clinical practice. Large language model (LLM)&#x2013;based systems may support patient-facing interpretation of CGM data, but evidence remains limited for retrieval-grounded tools evaluated against clinician-authored responses in counseling scenarios. The system was intended for CGM interpretation and communication support rather than autonomous therapeutic decision-making.</p></sec><sec><title>Objective</title><p>This study aimed to evaluate whether a retrieval-grounded LLM-based conversational agent (CA) could support patient understanding of CGM data and preparation for diabetes consultations by generating responses to questions arising during CGM-informed diabetes counseling, with quality comparable to clinician-authored responses.</p></sec><sec sec-type="methods"><title>Methods</title><p>We developed a scaffolded LLM-based CA for CGM interpretation and diabetes counseling support. The system was designed to provide plain-language explanations of CGM patterns and responses to diabetes management questions while avoiding directive or individualized medical advice, such as recommending medication initiation, dose adjustment, or regimen changes. Around 12 CGM-informed cases, each comprising a deidentified CGM trace, a synthetic patient vignette, and accompanying CGM visual materials, were constructed from using available clinical datasets. Between October 2025 and February 2026, 6 senior UK diabetes clinicians each reviewed 2 assigned cases and answered 24 questions (12 per case). In a source-masked multirater evaluation, each CA-generated and clinician-authored response was independently rated by 3 clinicians on 6 quality dimensions, including clinical accuracy, guideline adherence, actionability, personalization, communication clarity, and empathy. Safety flags and perceived source labels were also recorded. The primary analysis used linear mixed effects models with random intercepts for case and rater.</p></sec><sec sec-type="results"><title>Results</title><p>A total of 288 unique responses (144 CA and 144 clinician responses) were evaluated, generating 864 ratings. CA-generated responses received higher quality scores than clinician-authored responses under controlled vignette-based conditions, with mean scores of 4.37 (SD 0.57) versus 3.58 (SD 0.90) and an estimated mean difference of 0.782 points on a 5-point scale (95% CI 0.692&#x2010;0.872; <italic>P</italic>&#x003C;.001). This pattern was observed across all 6 categories of patient questions. The largest estimated differences were for empathy (mean difference 1.062, 95% CI 0.948&#x2010;1.177) and actionability (0.992, 95% CI 0.877&#x2010;1.106). Safety flag distributions were similar between CA and clinician responses, with major concerns rare in both groups (n=3, 0.7% each). Although CA responses were longer, additional analyses adjusting for word count did not indicate that response length explained the overall quality difference.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Scaffolded LLM-based systems may have value as adjunct tools for CGM review, patient education, and preconsultation preparation by supporting standardized explanatory tasks. However, these findings should be interpreted in light of the vignette-based design, restricted datasets, and a small clinician panel, and they do not establish suitability for autonomous therapeutic decision-making, medication adjustment, or unsupervised real-world use. Prospective validation in clinical workflows is needed before implementation.</p></sec></abstract><kwd-group><kwd>continuous glucose monitoring</kwd><kwd>large language model</kwd><kwd>retrieval augmented generation</kwd><kwd>conversational agent</kwd><kwd>patient-facing AI</kwd><kwd>clinical evaluation</kwd><kwd>diabetes care</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>Diabetes mellitus is a chronic metabolic disease characterized by hyperglycemia due to impaired insulin secretion, insulin action, or both, and is associated with substantial microvascular and macrovascular complications [<xref ref-type="bibr" rid="ref1">1</xref>]. The International Diabetes Federation (IDF) reports that in 2024, approximately 589 million adults aged 20&#x2010;79 years were living with diabetes worldwide (11.1% of the adult population), with this number projected to reach 853 million by 2050, an increase of 46% [<xref ref-type="bibr" rid="ref2">2</xref>]. Given this growing global burden, digital tools are increasingly being investigated to support day-to-day monitoring, interpretation of glycemic patterns, and communication around self-management [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref4">4</xref>].</p><p>Continuous glucose monitoring (CGM) has become central to contemporary diabetes care [<xref ref-type="bibr" rid="ref5">5</xref>]. CGM systems provide near real-time interstitial glucose readings and trend information, enabling assessment of time in range (TIR), glycemic variability, and hypoglycemia risk [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. Evidence from clinical trials and consensus reports indicates that CGM use can increase TIR, reduce hypoglycemia, and improve treatment satisfaction in both type 1 and type 2 diabetes [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. Recent American Diabetes Association (ADA) Standards of Care recommend CGM for most individuals on intensive insulin therapy and support broader adoption where feasible [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref9">9</xref>]. As device accuracy, wearability, and reimbursement have improved, CGM use in routine clinical practice has expanded rapidly [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>].</p><p>Despite these benefits, interpreting CGM data remains challenging for many people living with diabetes. Although uptake is increasing, access and sustained use remain affected by device cost, insurance coverage, and variation in prescribing practices across clinical settings [<xref ref-type="bibr" rid="ref11">11</xref>]. Modern CGM platforms increasingly provide structured summaries of glucose data for patient and clinical review, including pattern reports, trend visualizations, and summary statistics [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>]. Dexcom Clarity, for example, highlights glucose patterns, trends, and statistics through report-based summaries [<xref ref-type="bibr" rid="ref12">12</xref>], whereas Abbott&#x2019;s LibreView provides reports such as Glucose Pattern Insights that highlight glycemic patterns and may include medication and lifestyle considerations for review [<xref ref-type="bibr" rid="ref13">13</xref>]. However, these tools are primarily designed to summarize and visualize data rather than to provide conversational, patient-facing explanations of why patterns may be occurring or how they relate to daily behaviors, treatment routines, and patient concerns. Even among CGM users, limited structured training in how to interpret traces means that many still feel overwhelmed by the volume and complexity of CGM readouts and struggle to relate observed patterns to food intake, physical activity, medication timing, or other day-to-day behaviors [<xref ref-type="bibr" rid="ref14">14</xref>]. Patients may be unsure how to respond to recurring highs or lows, how to understand trends over time, or which questions should be brought to clinicians during review [<xref ref-type="bibr" rid="ref15">15</xref>-<xref ref-type="bibr" rid="ref17">17</xref>]. Limited consultation time further constrains opportunities for detailed, individualized explanation of CGM traces in standard clinical encounters [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. Together, these practical and educational barriers highlight a need for scalable approaches that can support clear, timely, and patient-facing interpretation of CGM data alongside routine care.</p><p>In addition to these interpretive challenges, many people living with diabetes experience substantial emotional burden. Meta-analyses suggest that approximately 10%&#x2010;15% of adults with diabetes have clinically significant depressive symptoms [<xref ref-type="bibr" rid="ref18">18</xref>]. Diabetes distress and depression are associated with poorer glycemic control, reduced adherence to self-management behaviors, and lower quality of life [<xref ref-type="bibr" rid="ref19">19</xref>]. In this context, support tools for CGM-related communication must address not only the technical explanation of glucose patterns but also the uncertainty, frustration, and anxiety that often accompany diabetes self-management. This does not necessarily require formal psychological intervention, but it does underscore the importance of clarity, reassurance, and empathic communication in patient-facing diabetes support.</p><p>Recent advances in large language models (LLMs) offer a potential way to address unmet informational and communication needs in diabetes care [<xref ref-type="bibr" rid="ref20">20</xref>]. LLMs can accept structured and unstructured inputs, including numerical summaries, clinical text, and free-text patient queries, and generate contextualized natural-language explanations [<xref ref-type="bibr" rid="ref21">21</xref>]. This makes them candidates for supporting tasks such as CGM interpretation, patient education, and communication of case-specific information in accessible language [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. Early evaluations and review articles report that, in many scenarios, LLM-based systems provide reasonably accurate health information and can generate responses that show elements of cognitive empathy, such as recognizing expressed emotions and offering supportive language in simulated patient interactions [<xref ref-type="bibr" rid="ref22">22</xref>-<xref ref-type="bibr" rid="ref25">25</xref>]. However, current LLMs in medicine remain limited by hallucinated or incomplete content, variable alignment with clinical guidelines, and challenges in transparently incorporating patient-specific structured data, which raises concerns about their safe use in clinical contexts [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref26">26</xref>]. Retrieval-augmented generation (RAG) has been proposed as one strategy to mitigate these limitations by grounding LLM outputs in curated, evidence-based sources and explicitly linking responses to both guideline documents and case-specific data such as CGM traces [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref26">26</xref>]. Empirical evidence on the performance of such systems in patient-specific, CGM-informed diabetes scenarios, however, remains limited.</p><p>Recent studies have applied LLM- or RAG-based conversational agents (CAs) to diabetes care, mainly for diabetes education, lifestyle, and self-management advice, or general question and answering [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref28">28</xref>]. These systems are commonly evaluated in terms of diabetes knowledge, perceived usefulness, usability, and the appropriateness or safety of chatbot advice [<xref ref-type="bibr" rid="ref27">27</xref>-<xref ref-type="bibr" rid="ref30">30</xref>], but they rarely incorporate CGM data directly or examine how such systems perform when asked to interpret CGM-informed cases in a patient-facing manner. In parallel, other studies have used LLMs to analyze CGM data by calculating standard CGM metrics and generating narrative summaries of glucose traces, which endocrinologists then assess for accuracy, completeness, and safety, often using simulator-generated rather than real-world data [<xref ref-type="bibr" rid="ref31">31</xref>]. Collectively, these studies suggest that LLM-based tools may assist with diabetes-related information provision and CGM analysis [<xref ref-type="bibr" rid="ref27">27</xref>-<xref ref-type="bibr" rid="ref31">31</xref>]. However, they have not adequately evaluated retrieval-grounded LLM-based CAs in CGM-informed scenarios that require structured glucose interpretation together with patient-centered communication, nor have they compared specialist ratings of CA-generated responses against clinician-authored responses to the same cases.</p><p>In practice, one plausible role for such a system may be as an adjunct to routine diabetes care, particularly for preconsultation preparation, standardized explanation of CGM patterns, and support for common patient-facing questions that are structured but time-consuming to address repeatedly in routine care [<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>]. Before a scheduled review, patients might use such a system to obtain a plain-language explanation of recent glucose patterns, identify recurring concerns, and formulate questions for discussion with their diabetes care team. In this way, the system could potentially help surface common issues in advance, support more consistent handling of routine CGM-related questions, and improve consultation efficiency by allowing clinicians to focus more directly on individualized decision-making and more complex cases. More broadly, this suggests a possible practical role for LLM-based tools in diabetes services, where many routine advisory tasks involve synthesizing structured and semistructured information into clear, context-specific communication, while final treatment recommendations and higher-risk clinical judgments remain clinician-led. Accordingly, the potential value of such systems may lie less in autonomous decision-making than in supporting routine explanatory work, augmenting specialist capacity, and helping prioritize clinician time toward situations requiring greater clinical complexity or safety oversight [<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref34">34</xref>].</p></sec><sec id="s1-2"><title>Study Aim</title><p>To our knowledge, this study is among the first to conduct a source-masked comparative evaluation in which diabetes specialists assessed both retrieval-grounded LLM-generated and clinician-authored responses to identical CGM-informed cases across predefined clinical and psychosocial evaluation criteria. To address the need for scalable support in patient-facing communication around CGM review, we developed a scaffolded LLM GPT-5.1&#x2013;based LLM CA incorporating RAG that generated structured, plain-language responses to questions arising in CGM-informed diabetes counseling using case-specific CGM information, vignette context, and guideline-based retrieval. We then conducted a source-masked multirater evaluation between October 2025 and February 2026 with UK diabetes clinicians, who assessed CA- and clinician-authored responses for quality and safety. In doing so, this study examined whether such a system could support standardized explanatory and educational aspects of diabetes care in consultations involving CGM review, without positioning it as a substitute for clinician judgment.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Ethical Considerations</title><p>This study did not involve direct contact with patients or access to identifiable clinical records. The CGM files were derived from publicly available, preexisting, deidentified clinical datasets, and all accompanying patient vignettes were synthetic and created solely for research purposes. The human contributors were 6 diabetes clinicians who served as expert reviewers by providing written responses and rating anonymized text outputs. They were not asked to provide personal or sensitive information and were considered expert assessors rather than research participants. According to UK Health Research Authority guidance, research using publicly available or fully anonymized data in which individuals cannot be identified may not require formal Research Ethics Committee review [<xref ref-type="bibr" rid="ref35">35</xref>]. Because this study used publicly available anonymized CGM datasets together with clinician ratings of generated responses and did not involve identifiable patient information, formal ethics approval was not required.</p></sec><sec id="s2-2"><title>CGM Data and Case Vignettes</title><p>An overview of the complete workflow is presented in <xref ref-type="fig" rid="figure1">Figures 1A-1D</xref>. We constructed 12 CGM-based cases using deidentified glucose traces from 2 publicly available research datasets. Six traces (3 type 1 diabetes and 3 type 2 diabetes) were sampled from the ShanghaiT1DM and ShanghaiT2DM datasets [<xref ref-type="bibr" rid="ref36">36</xref>], which provide 3&#x2010;14 days of CGM values at 15-minute intervals for 12 individuals with type 1 diabetes and 100 individuals with type 2 diabetes, together with capillary blood glucose measurements, blood ketone, self-reported dietary intake, insulin doses, and clinical characteristics [<xref ref-type="bibr" rid="ref36">36</xref>]. For the 6 Shanghai cases used in this study, the CGM monitoring period ranged from 7 to 14 days: 1 case included 7 consecutive days of data, and the remaining 5 included at least 10 days of CGM readings. In addition to the CGM glucose time series, selected contextual fields from the source datasets, including dietary intake and insulin records, were used to inform the construction of the synthetic patient vignettes. These variables were used only for case contextualization and were not analyzed as separate study variables.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Overview of the study workflow. (<bold>A</bold>) Construction of 12 deidentified continuous glucose monitoring (CGM) cases and corresponding synthetic patient vignettes. (<bold>B</bold>) Retrieval-grounded conversational agent (CA) system using GPT-5.1, structured CGM summaries, vignette context, and guideline excerpts retrieved through retrieval-augmented generation (RAG). (<bold>C</bold>) Two-phase clinician study involving response generation and source-masked rating. (<bold>D</bold>) Statistical analyses, including interrater reliability assessment, mixed effects models, sensitivity analyses, domain-specific analyses, and source-identification analyses. CA: conversational agent; CGM: Continuous glucose monitoring; FAISS: Facebook AI similarity search; ICC: intraclass correlation coefficient.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e98519_fig01.png"/></fig><p>The remaining 6 traces were selected from the OhioT1DM 2018 dataset [<xref ref-type="bibr" rid="ref37">37</xref>], which contains 8 weeks of CGM, insulin pump, physiological sensor, and self-reported life-event data for 6 adults with type 1 diabetes [<xref ref-type="bibr" rid="ref37">37</xref>]. In addition to the CGM glucose measurements, selected self-reported life event information was used to inform the construction of the synthetic patient vignettes. As with the Shanghai cases, these variables were used only for case contextualization and were not analyzed as separate study variables. The resulting case set comprised 9 type 1 diabetes cases and 3 type 2 diabetes cases. This distribution reflected both the composition of the source datasets and the more limited availability of suitable publicly available type 2 diabetes CGM datasets for case construction, while allowing the evaluation to include both major diabetes types and a range of CGM pattern presentations. Within the type 2 diabetes group, cases were not further selected to represent specific treatment regimens, such as insulin-treated or noninsulin-treated subgroups. Instead, the selected cases should be interpreted as examples of CGM-informed counseling scenarios in type 2 diabetes rather than as representing the full clinical heterogeneity of the broader type 2 diabetes population.</p><p>To approximate a realistic consultation context while preserving privacy, we created a synthetic case vignette for each selected CGM trace. Each vignette included a brief clinical profile, such as age band, sex, diabetes type and duration, treatment regimen, typical diet and activity patterns, and the patient&#x2019;s main concerns about glycemic control, together with one or more CGM plots (an example can be found in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) generated from the underlying trace to provide clinicians with an intuitive visual overview of glucose patterns. For both datasets, the CGM glucose traces provided the primary basis for case selection, CGM metric calculation, and visual material generation, whereas available contextual fields, including dietary intake, insulin records, treatment information, clinical characteristics, and self-reported life-event details, were used only as anchors for constructing synthetic vignettes and were not analyzed as separate study variables. Where such fields were incomplete, details were synthetically expanded only to provide plausible counseling context. Coherence was assessed by checking consistency with diabetes type, treatment modality, CGM summary metrics, and visible glucose patterns, including TIR, time above range (TAR), time below range (TBR), glycemic variability, nocturnal lows, postprandial excursions, fasting or early-morning rises, and data completeness. Draft vignettes and plots were prepared by ZG and refined through formative review with senior diabetologist EK; refinements were limited to improving clinical plausibility, differentiating type 1 and type 2 diabetes contexts, and representing early-morning glucose rise or possible dawn phenomenon where supported by the CGM trace.</p></sec><sec id="s2-3"><title>CA Design</title><p>A retrieval-grounded CA for CGM interpretation and diabetes-related counseling scenarios was implemented using the GPT-5.1 LLM accessed via the OpenAI API. GPT-5.1 was one of the most advanced models available at the time of system development; no comparative benchmarking across alternative models was undertaken. The base model was used without parameter fine-tuning. Task adaptation relied on structured prompting combined with RAG. Deidentified CGM summaries and vignette-based patient queries were embedded into predefined prompt templates, and guideline-aligned reference materials were retrieved from a curated knowledge base to condition response generation.</p><p>For each case, the CGM glucose time series was processed in Python using a custom analysis function that calculated core consensus metrics, including monitoring duration, data completeness, mean glucose, SD, coefficient of variation, and the percentage of readings within standard CGM ranges (TIR 70&#x2010;180 mg/dL [3.9&#x2010;10.0 mmol/L], TBR &#x003C;70 mg/dL [&#x003C;3.9 mmol/L] and &#x003C;54 mg/dL [&#x003C;3.0 mmol/L], and TAR &#x003E;180 mg/dL [&#x003E;10.0 mmol/L] and &#x003E;250 mg/dL [&#x003E;13.9 mmol/L]), consistent with international recommendations for CGM-derived metrics and TIR reporting [<xref ref-type="bibr" rid="ref6">6</xref>]. The function generated a concise text summary listing these metrics alongside commonly used target thresholds. This standardized CGM summary was used as structured input to the CA.</p><p>To verify the correctness of the implementation, ZG manually recalculated the CGM metrics for representative traces from both the Shanghai and Ohio datasets using Microsoft Excel as an implementation check. For these validation traces, all metrics were recomputed using simple count and average formulas based on the international CGM TIR consensus definitions [<xref ref-type="bibr" rid="ref38">38</xref>], and the Excel results exactly matched the Python (Python Software Foundation) outputs at the precision reported. This validation step was performed solely to confirm the correctness of the implementation and to support basic quality checks, such as visual inspection, so that the reported metrics were consistent with the plotted CGM traces. The metric calculations themselves were not shown to clinician raters. For each question, the model input consisted of the CGM summary, the synthetic case vignette (clinical profile and contextual information), the text of a prespecified question from the case-specific question set, and image-based CGM visualizations for the case, including both the continuous glucose trace and an ambulatory glucose profile (AGP)-style profile. These visualizations were supplied directly to the CA as multimodal inputs, so the model&#x2019;s interpretation was informed by both structured summary metrics and visual glucose pattern information. An example CGM summary format is provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>To ground responses in current evidence and professional guidance, a RAG approach was used. A domain-specific corpus was built from publicly available materials on CGM metrics and interpretation [<xref ref-type="bibr" rid="ref38">38</xref>-<xref ref-type="bibr" rid="ref41">41</xref>], CGM-guided glucose management [<xref ref-type="bibr" rid="ref42">42</xref>-<xref ref-type="bibr" rid="ref46">46</xref>], AGP reports [<xref ref-type="bibr" rid="ref39">39</xref>], major diabetes care guidelines [<xref ref-type="bibr" rid="ref47">47</xref>,<xref ref-type="bibr" rid="ref48">48</xref>], and resources on emotional and psychosocial aspects of diabetes [<xref ref-type="bibr" rid="ref47">47</xref>-<xref ref-type="bibr" rid="ref51">51</xref>]. The corpus also included selected segments from the MedDialog 2020 medical dialogue dataset to provide examples of clinician-patient conversational style [<xref ref-type="bibr" rid="ref52">52</xref>] and curated patient education materials from national health services and diabetes charities (such as Diabetes UK [<xref ref-type="bibr" rid="ref50">50</xref>] and the ADA [<xref ref-type="bibr" rid="ref51">51</xref>]), which were manually organized into topic-based summaries. All documents were converted to plain text and split into short overlapping segments using a recursive character-based splitter with a target segment length of approximately 500 characters and an overlap of 100 characters. Each segment was embedded using OpenAI&#x2019;s text embedding service via the LangChain OpenAIEmbeddings wrapper [<xref ref-type="bibr" rid="ref53">53</xref>] and stored in a Facebook AI Similarity Search (FAISS) vector index implementing approximate nearest-neighbor search [<xref ref-type="bibr" rid="ref54">54</xref>]. At inference time, the current question text was used as the retrieval query to obtain the top 2 nearest segments from the FAISS index. The retrieval function returned a Euclidean (L2) distance score, with lower values indicating closer embedding-space proximity.</p><p>Two prompt templates were used to structure model behavior across interaction stages. The first template was used to generate an initial plain-language explanation of the precomputed CGM summary and vignette context, with retrieved guideline-aligned reference materials incorporated through the RAG pipeline to support interpretation. The CGM metrics themselves were computed in Python in advance and provided to the model as structured input rather than being generated by the prompt. The second template was used to answer follow-up patient questions, again incorporating retrieved reference materials through the RAG pipeline to support context-aware responses. Full prompt templates are provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. The system prompt defined the CA as a diabetes specialist and instructed it to provide clear, plain-language, empathetic, and nonjudgmental explanations aligned with standard diabetes education practices. During system development, selected example outputs were reviewed by a clinical collaborator (EK) to provide targeted feedback on tone, clarity, and clinical appropriateness, and this feedback informed the final prompt refinement. For each case, question pair in the formal evaluation, the model produced a single free-text response without iterative refinement.</p><p>For formal evaluation, all CA responses were generated on November 14, 2025, using the OpenAI API model identifier gpt-5.1. A fixed configuration was applied to all formal case-question pairs. Generation parameters were set to a temperature of 0.1, a top_p value=1.0, and maximum output length of 1200 tokens. Frequency and presence penalties were left at default settings, no stop sequences were specified, and no seed parameter was specified in the API call. The low temperature setting was selected to reduce output variability and limit speculative generation [<xref ref-type="bibr" rid="ref22">22</xref>]. API credentials were stored as environment variables and were not embedded in the code or shared materials. For retrieval, documents were embedded using text-embedding-3-small, split into approximately 500-character chunks with 100-character overlap, indexed in FAISS, and retrieved with top_k=2 using the current question text as the retrieval query.</p></sec><sec id="s2-4"><title>Safety and Security</title><p>Several safeguards were implemented to minimize the risk of inappropriate, misleading, or clinically unsafe outputs. Inputs to the CA were restricted to deidentified CGM summaries and synthetic vignettes, so no identifiable patient information or directly actionable treatment instructions were supplied. Behavioral constraints embedded in the system prompted the CA to avoid prescriptive therapeutic recommendations and to encourage consultation with the person&#x2019;s usual diabetes care team when glycemic patterns suggested potential risk, reinforcing its role as a communication support tool for structured CGM explanation rather than a substitute for clinical judgment. The RAG component was designed to expose the model to a curated, guideline-consistent corpus rather than unrestricted external sources, supporting retrieval traceability and potential guideline alignment; all CGM metrics used to contextualize model reasoning were validated in advance to avoid propagating incorrect numerical information. During system development, draft outputs, for example, case-question pairs, were iteratively reviewed by diabetes clinicians, and their feedback was used to refine the prompt and guardrails until the response style was judged acceptable for the study. Potential failure modes considered during system design included misinterpretation of CGM patterns, overgeneralized lifestyle advice, excessive reassurance, inappropriate specificity in medication-related responses, failure to recognize clinically concerning hypoglycemia or hyperglycemia patterns, and mismatch with local prescribing or eligibility criteria. These risks were mitigated through prompt-based safety constraints, retrieval from a curated guideline-consistent corpus, clinician review during development, and the use of safety flags in the formal evaluation.</p></sec><sec id="s2-5"><title>Study Design</title><p>This study used a vignette-based, multirater, source-masked comparative design to evaluate the CA&#x2019;s responses against those of diabetes clinicians. The expert panel comprised 6 senior diabetes specialists recruited from major UK National Health Service (NHS) Foundation Trusts, including Imperial College Healthcare NHS Trust (St Mary&#x2019;s Hospital), University College London Hospitals NHS Foundation Trust (UCLH), and the Royal Free London NHS Foundation Trust. The use of senior diabetes specialists provided a stringent benchmark for comparison.</p><p>Participants met predefined eligibility criteria: (1) a primary clinical appointment within a tertiary referral center or major teaching hospital; (2) at least 10 years of postqualification clinical experience; and (3) demonstrated expertise in diabetes technology, particularly in the clinical interpretation of CGM and advanced insulin delivery systems. The panel included consultant diabetologists and senior specialists, several of whom hold clinical-academic appointments.</p><p>After constructing 12 CGM-based cases with synthetic clinical vignettes as described above, 6 UK diabetes clinicians were first invited to provide written responses and subsequently to rate anonymized response pairs. Each clinician was randomly assigned 2 of the 12 cases and received the full standardized case package for each assigned case. This included the synthetic vignette together with structured supporting materials, including demographic and clinical background, current treatment, CGM summary metrics, lifestyle and self-management information, patient-reported observations, accompanying CGM visual materials, and the raw CGM data.</p><p>The question set was grouped into 6 content domains reflecting topics commonly discussed in diabetes consultations in which CGM review forms an important component: (1) blood glucose interpretation and fluctuation analysis (6 questions), (2) impact of food and exercise (7 questions), (3) medication and treatment guidance (4 questions), (4) emotional and psychological concerns (5 questions), (5) long-term goals and motivation (5 questions), and (6) technical issues and device use (4 questions). These domains were selected to reflect the breadth of patient-facing questions that may arise in such consultations, ranging from direct interpretation of glucose patterns to treatment-related, psychosocial, goal-oriented, and device-related concerns [<xref ref-type="bibr" rid="ref29">29</xref>]. The draft question bank was reviewed by senior diabetologist EK, who refined its content and added questions based on issues commonly encountered in routine diabetes consultations. The domains were not intended to depend equally on CGM data: some required direct interpretation of CGM metrics and trends, whereas others drew more on the broader clinical context in which CGM findings are discussed. The full question bank is provided in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p><p>For each assigned case, clinicians were asked to answer 12 questions drawn from this bank. To ensure coverage of clinically salient topics, each case included 3 questions from the blood glucose interpretation and fluctuation analysis domain, 3 questions related to the impact of food and exercise, 2 questions on medication and treatment guidance, 2 questions on emotional and psychological concerns, and one question each on long-term goals and motivation and on technical issues and device use. Questions specific to type 1 diabetes or intensive insulin therapy (for example, detailed insulin adjustment questions) were not allocated to type 2 diabetes cases. Across the 12 cases, individual question templates were reused with different vignettes and assigned to different clinicians so that each item was answered in multiple clinical contexts while minimizing overlap in the exact question sets seen by any single clinician. Assignment details and frequency statistics for each question are provided in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>.</p><p>In the first phase, all case materials were distributed to clinicians on October 22, 2025, and all clinician-authored responses were received by November 18, 2025. Clinicians were instructed to answer their allocated questions as they would when writing to a patient in routine practice, using plain English and making reasonable assumptions based only on the CGM information and vignette provided. No formal time limit was imposed, but clinicians were asked to respond in a manner consistent with their usual clinical communication style. For each case and question, clinicians entered a single free-text response.</p><p>To promote broadly comparable levels of detail while preserving individual clinical expression, clinicians were provided with a nonmandatory guideline suggesting responses of approximately 150&#x2010;250 words, with flexibility according to content complexity. No strict word limit was imposed. Final CA responses were generated independently on November 14, 2025, using the locked case materials, question bank, prompt templates, RAG corpus, and prespecified generation configuration. Although some clinician-authored responses had been received by that date, they were not used in any part of CA response generation, including model training, fine-tuning, prompt construction, retrieval corpus development, output comparison, or output revision. The CA did not have access to clinician-authored responses during generation. The use of GPT-5.1 reflected the intention to evaluate the most advanced model available at the time of final CA response generation without changing the underlying case materials, prompts, retrieval corpus, or evaluation procedure. The CA generated one free-text response per case-question pair under the same configuration. No explicit word-count constraint was imposed on CA outputs; response length was determined by the model&#x2019;s generation process within the API token limit. All CA outputs and clinician responses were exported and stored as separate, anonymized text units, labeled only by case ID and question ID.</p><p>The second phase was assigned to clinicians on November 19, 2025, after both clinician-authored and CA-generated response sets had been prepared, and all ratings were received by February 15, 2026. In this phase, the same 6 clinicians rated anonymized responses without being informed of the response source. For each case-question pair, raters were provided with the corresponding original case materials and an anonymized response set that included the CA answer and clinician-authored answers written by other clinicians. Raters did not evaluate their own responses. Source recognizability was assessed empirically through the perceived-source label.</p><p>Several design features were used to reduce direct self-rating and single-rater calibration effects. First, each response was rated by 3 clinicians rather than by a single assessor, so quality estimates reflected multirater judgment rather than one individual clinician&#x2019;s scoring pattern. Second, clinicians were recruited from multiple UK NHS Trusts and were not informed which other clinicians had contributed comparator responses, reducing the likelihood of coordination or recognition of colleague-authored responses. Third, responses were anonymized and labeled only by case and question identifiers during rating. Nevertheless, because the same specialist panel contributed comparator responses in Phase 1 and later rated anonymized responses in Phase 2, the rating panel should not be regarded as fully independent of the comparator-generation process.</p><p>During the review process, clarification was also provided to all raters regarding the interpretation of certain evaluation criteria, particularly guideline adherence and actionability, when applied to more explanatory questions. Specifically, raters were advised that guideline adherence should be judged in terms of consistency with appropriate clinical practice even when no guideline was explicitly cited and that actionability should not penalize otherwise strong explanatory responses when the question did not naturally call for specific next-step advice.</p><p>Quality was rated on 6 5-point Likert-scale dimensions (1=very poor-5=excellent), including clinical accuracy, guideline adherence, actionability, personalization, communication clarity, and empathy/emotional support. The operational definitions of each quality dimension and the rating scale are summarized in <xref ref-type="table" rid="table1">Table 1</xref>. Safety was captured with a 3-level flag (0=no safety concerns, 1=minor concern requiring revision before clinical use, and 2=major concern with unsafe or clearly contraindicated advice), and raters were also asked to guess the likely source of each response (&#x201C;human clinician,&#x201D; &#x201C;LLM,&#x201D; or &#x201C;not sure&#x201D;; <xref ref-type="table" rid="table1">Table 1</xref>). Each response, therefore, received up to 3 sets of quality scores, one safety flag, and one perceived source label for subsequent analysis.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Quality rating dimensions, safety flag categories, and perceived source labels.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Item</td><td align="left" valign="bottom">Definition / what raters were asked to consider</td><td align="left" valign="bottom">Scale/categories</td></tr></thead><tbody><tr><td align="left" valign="top">Clinical accuracy</td><td align="left" valign="top">Correctness of CGM<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> interpretation and use of numerical information (eg, TIR<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup>/TBR<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup>/TAR<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup>, mean glucose, variability indices).</td><td align="left" valign="top">5-point Likert scale (1=very poor-5=excellent)</td></tr><tr><td align="left" valign="top">Guideline adherence</td><td align="left" valign="top">Alignment with established diabetes and CGM practice, including consistency with major CGM and diabetes care guidelines.</td><td align="left" valign="top">5-point Likert scale (1=very poor-5=excellent)</td></tr><tr><td align="left" valign="top">Actionability</td><td align="left" valign="top">Clarity and feasibility of suggested next steps and contingencies for the patient, given the CGM patterns and vignette context.</td><td align="left" valign="top">5-point Likert scale (1=very poor-5=excellent)</td></tr><tr><td align="left" valign="top">Personalization</td><td align="left" valign="top">Explicit use of case-specific details from the vignette and CGM data, rather than generic or template-like advice.</td><td align="left" valign="top">5-point Likert scale (1=very poor-5=excellent)</td></tr><tr><td align="left" valign="top">Communication clarity</td><td align="left" valign="top">Ease of understanding for a layperson, including structure, wording, and avoidance of jargon or ambiguous phrasing.</td><td align="left" valign="top">5-point Likert scale (1=very poor-5=excellent)</td></tr><tr><td align="left" valign="top">Empathy / emotional support</td><td align="left" valign="top">Degree of validation, nonjudgmental tone, acknowledgment of emotional burden, and provision of supportive, encouraging language.</td><td align="left" valign="top">5-point Likert scale (1=very poor-5=excellent)</td></tr><tr><td align="left" valign="top">Safety flag</td><td align="left" valign="top">Presence and severity of safety concerns in the advice provided (eg, unsafe insulin suggestions, failure to respond to very high or low glucose).</td><td align="left" valign="top">0=no safety concerns; 1=minor concern requiring revision before clinical use; 2=major concern with unsafe or clearly contraindicated advice</td></tr><tr><td align="left" valign="top">Perceived source</td><td align="left" valign="top">Rater&#x2019;s guess about whether the response was written by a human clinician or by the CA<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup>.</td><td align="left" valign="top">&#x201C;Human clinician,&#x201D; &#x201C;LLM<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup>,&#x201D; or &#x201C;Not sure&#x201D;</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>CGM: continuous glucose monitoring.</p></fn><fn id="table1fn2"><p><sup>b</sup>TIR: time in range.</p></fn><fn id="table1fn3"><p><sup>c</sup>TBR: time below range.</p></fn><fn id="table1fn4"><p><sup>d</sup>TAR: time above range.</p></fn><fn id="table1fn5"><p><sup>e</sup>CA: conversational agent.</p></fn><fn id="table1fn6"><p><sup>f</sup>LLM: large language model.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-6"><title>Post Hoc Retrieval Audit</title><p>Twenty-four case-question pairs were audited, with 4 responses selected from each of the 6 predefined question domains. The audited items were selected using a domain-stratified sequential sampling approach: case-question pairs were reviewed in case order, and items were selected in a rotating manner across domains 1-5 to maximize coverage of the predefined question domains, case IDs, and question IDs. Selection was not based on retrieval score, response quality, or the apparent strength of retrieval-response alignment. The audit was therefore intended to provide structured coverage and transparency across the question taxonomy, rather than a statistically representative estimate of retrieval-response alignment across the full response set.</p><p>For each audited item, the retrieval process was reconstructed using the same RAG corpus, recursive character-based chunking strategy, embedding model, FAISS vector index, and top k retrieval setting used in the original system configuration. The audit recorded the case ID, question ID, domain, top-ranked source document, retrieved text segment, archived CA response excerpt, FAISS retrieval score, manual retrieval-response alignment category, evidence-to-response role, and rationale for classification. The full item-level audit table is provided in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>.</p><p>The retrieval score was the FAISS L2 distance returned by the similarity search function. Lower values indicate greater embedding space proximity between the query and the retrieved segment [<xref ref-type="bibr" rid="ref55">55</xref>]. This score was used as a quantitative trace of the retrieval process and as a descriptive indicator of why a segment was returned by vector search. It was not treated as a clinical relevance score or as evidence that the retrieved segment necessarily influenced the final CA response. Because embedding space proximity does not directly establish clinical relevance or response grounding, retrieval-response relevance was assessed separately through structured manual review.</p><p>Two authors (ZG and KL) independently reviewed the retrieved segment, original question, and archived CA response excerpt for each audited item using a predefined rubric. Retrieved segments were classified as directly aligned, broadly aligned, weakly aligned, or showing no clear alignment with the CA response. Directly aligned indicated that the retrieved text explicitly supported the main claim, explanation, or recommendation in the CA response. Broadly aligned indicated that the retrieved text was relevant to the same clinical topic but provided general contextual support rather than direct evidence for a specific response element. Weakly aligned indicated only indirect or tangential relevance. No clear alignment indicated that no meaningful relationship between the retrieved segment and the question or CA response was visible from the audited materials.</p><p>To assess consistency between reviewers, the 4 alignment categories were coded ordinally from 0 to 3, with no clear alignment coded as 0, weakly aligned as 1, broadly aligned as 2, and directly aligned as 3. Interreviewer agreement was assessed using an intraclass correlation coefficient. The agreement between ZG and KL was good (intraclass correlation coefficient [ICC]=0.875). Disagreements were resolved through adjudication by a third author (AL). The apparent role of retrieved evidence was also categorized as directly reflected in the response, conceptually consistent, background or context only, or no clear contribution. This audit was intended to assess retrieval traceability and response alignment, rather than to establish a causal effect of retrieval on response quality.</p></sec><sec id="s2-7"><title>Statistical Analysis</title><p>Analyses were based on the clinician ratings described above. For each response, 3 raters provided 6 1&#x2010;5 quality scores, a 3-level safety flag, and a perceived source label. An overall quality score was defined as the arithmetic mean of the 6 quality items for a given rating. Quality scores for CA and clinician responses were summarized using means, SDs, medians, and IQRs, overall and by question domain. Heatmaps were generated to visualize mean quality ratings across specific domains and dimensions.</p><p>Interrater reliability for the quality ratings was evaluated using 2-way random-effects ICCs for absolute agreement [<xref ref-type="bibr" rid="ref56">56</xref>]. Both single-rater (ICC(2,1)) and average-rater (ICC(2,3)) reliabilities were calculated separately for each quality item and for the overall quality score, with the 3 raters per response treated as interchangeable observers. The 95% CIs were estimated using cluster bootstrap resampling [<xref ref-type="bibr" rid="ref57">57</xref>] at the response level.</p><p>To compare CA and clinician performance, the 1&#x2010;5 quality ratings were treated as approximately continuous, consistent with common practice for Likert-type scales in health services research. The primary analysis used linear mixed effects models fitted separately for each quality item and for the overall quality score [<xref ref-type="bibr" rid="ref58">58</xref>]. In these models, the individual rating was the outcome, responder type (CA vs clinician) was included as a fixed effect with the clinician set as the reference category, and random intercepts for case and rater were included to account for clustering of ratings within CGM cases and systematic differences in individual clinicians&#x2019; scoring tendencies. The rater-level random intercept was included to account for potential differences in rating calibration between clinicians, including differences that may have arisen from their clinical background, communication norms, or prior experience of authoring comparator responses in Phase 1. Results were reported as estimated mean differences (CA minus clinician) with 95% CIs and 2-sided <italic>P</italic> values [<xref ref-type="bibr" rid="ref59">59</xref>]. As a sensitivity analysis, the 3 rater scores for each response were first averaged to create matched case-question pairs. The normality of the paired differences was assessed using the Shapiro-Wilk test [<xref ref-type="bibr" rid="ref60">60</xref>]. Because the differences deviated from a normal distribution, comparisons between CA and clinician mean scores were conducted using the nonparametric Wilcoxon signed-rank test [<xref ref-type="bibr" rid="ref61">61</xref>].</p><p>To examine whether response length was associated with quality ratings, word count was included as a continuous fixed-effect covariate in additional linear mixed effects models [<xref ref-type="bibr" rid="ref58">58</xref>]. These models included random intercepts for unique response ID and rater to account for repeated ratings of the same response and systematic differences in rater calibration. An interaction term between responder type and word count was incorporated to assess whether the association between length and quality differed between CA and clinician responses. Statistical significance of the main and interaction effects was evaluated using Wald tests [<xref ref-type="bibr" rid="ref62">62</xref>].</p><p>Domain-specific analyses were conducted to explore whether relative performance varied by clinical topic. For each of the 6 predefined domains, the mixed effects models were refitted on the subset of responses belonging to that domain. Because these analyses involved multiple comparisons and the study was not primarily powered for domain-level hypotheses, domain-specific results are interpreted as exploratory, with an emphasis on effect sizes and CIs rather than formal adjustment for multiplicity.</p><p>Safety flags were tabulated for CA and clinician responses and are reported descriptively, given the very low number of flagged responses. For the source-identification task (Turing test [<xref ref-type="bibr" rid="ref63">63</xref>]), the proportions of responses that raters correctly identified were calculated both overall and at the individual rater level. Identification accuracy was compared against random chance (50%) using exact binomial tests [<xref ref-type="bibr" rid="ref64">64</xref>]. Furthermore, to assess potential evaluation bias, overall quality scores were stratified and visualized using boxplots based on the raters&#x2019; perceived source (ie, whether the rater believed the response was generated by a CA or a clinician), regardless of the true source.</p><p>Missing ratings were rare and handled using complete-case analysis without imputation. All statistical tests were 2-sided (except for the one-sided binomial tests assessing accuracy greater than chance), with a significance threshold of <italic>P</italic>&#x003C;.05. Analyses were conducted using Python 3.11.</p></sec><sec id="s2-8"><title>Reporting Framework</title><p>This study was a simulated, vignette-based early-stage evaluation of an AI-enabled decision-support and communication-support system, rather than a prospective interventional trial protocol or a randomized clinical trial. To support transparent and reproducible reporting of the AI system and its evaluation, we used the Developmental and Exploratory Clinical Investigations of Decision Support Systems Driven by AI (DECIDE-AI) reporting guideline [<xref ref-type="bibr" rid="ref65">65</xref>]. DECIDE-AI was developed to guide early-stage clinical evaluations of AI-based decision support systems and was therefore aligned with this study design. The checklist was used to guide reporting of the system&#x2019;s intended use, intended users, deployment context, input data, model outputs, human oversight requirements, safety boundaries, and potential failure modes. The completed DECIDE-AI checklist is provided in <xref ref-type="supplementary-material" rid="app10">Checklist 1</xref>.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Rater Characteristics and Interrater Reliability</title><p>Six senior diabetes clinicians (male 4 and female 2), each with more than 10 years of postqualification clinical experience, participated in the rating process. A total of 12 CGM-informed diabetes cases were evaluated by the 6 clinicians. Each case comprised 12 structured questions. For every case-question unit, both a CA-generated response and a clinician-authored response were assessed.</p><p>In total, 288 unique case-question responses (144 CA and 144 clinician responses) were independently evaluated. Each response was rated by exactly 3 clinicians in a source-masked manner, yielding 864 response-level ratings. Each rating included 6 predefined quality dimensions (clinical accuracy, guideline adherence, actionability, personalization, communication clarity, and empathy), which were averaged to derive an overall quality score for analysis. The rating design was partially crossed but fully balanced at the response level, with no missing data. Details of clinician case assignments are provided in the Methods and <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>.</p><p>Interrater reliability (<xref ref-type="table" rid="table2">Table 2</xref>) was assessed using 2-way random-effects ICCs for absolute agreement. For the overall quality score, single-rater reliability was fair (ICC(1,2)=0.272; 95% CI 0.193&#x2010;0.342) and increased to a moderate level when averaging 3 raters (ICC(2,3)=0.529; 95% CI 0.418&#x2010;0.609). These findings supported the use of aggregated multirater scores for the primary response-level summaries, while indicating variability in individual clinician ratings. Exploratory subgroup analyses examined interrater reliability separately for CA-authored and clinician-authored responses. For overall quality, agreement was numerically higher for CA-authored responses than for clinician-authored responses at both the single-rater level (ICC(1,2) 0.154 vs 0.045; ICC difference 0.109, 95% CI &#x2013;0.066 to 0.159; <italic>P</italic>=.09) and the averaged-rater level (ICC(2,3) 0.354 vs 0.125; ICC difference 0.229, 95% CI &#x2013;0.137 to 0.258; <italic>P</italic>=.69), although neither difference was statistically supported.</p><p>Across individual quality dimensions, single-rater reliability was generally low to fair, ranging from 0.087 (clinical accuracy; 95% CI 0.007&#x2010;0.159) to 0.324 (actionability and empathy; 95% CI 0.245&#x2010;0.400 and 0.243&#x2010;0.405, respectively). Reliability improved consistently when ratings were averaged across 3 clinicians, with ICC(2,3) values ranging from 0.223 (clinical accuracy; 95% CI 0.022&#x2010;0.361) to 0.590 (actionability and empathy; 95% CI 0.493&#x2010;0.666 and 0.491&#x2010;0.671, respectively).</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Interrater reliability of clinician quality ratings.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Quality dimension</td><td align="left" valign="bottom" colspan="2">ICC<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>(2,1) (95% CI)</td><td align="left" valign="bottom" colspan="2">ICC(2,3) (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">Clinical accuracy</td><td align="left" valign="top" colspan="2">0.087 (0.007&#x2010;0.159)</td><td align="left" valign="top" colspan="2">0.223 (0.022&#x2010;0.361)</td></tr><tr><td align="left" valign="top">Guideline adherence</td><td align="left" valign="top" colspan="2">0.102 (0.023&#x2010;0.181)</td><td align="left" valign="top" colspan="2">0.254 (0.065&#x2010;0.399)</td></tr><tr><td align="left" valign="top">Actionability</td><td align="left" valign="top" colspan="2">0.324 (0.245&#x2010;0.400)</td><td align="left" valign="top" colspan="2">0.590 (0.493&#x2010;0.666)</td></tr><tr><td align="left" valign="top">Personalization</td><td align="left" valign="top" colspan="2">0.219 (0.141&#x2010;0.290)</td><td align="left" valign="top" colspan="2">0.457 (0.330&#x2010;0.551)</td></tr><tr><td align="left" valign="top">Clarity</td><td align="left" valign="top" colspan="2">0.214 (0.140&#x2010;0.292)</td><td align="left" valign="top" colspan="2">0.449 (0.328&#x2010;0.553)</td></tr><tr><td align="left" valign="top">Empathy</td><td align="left" valign="top" colspan="2">0.324 (0.243&#x2010;0.405)</td><td align="left" valign="top" colspan="2">0.590 (0.491&#x2010;0.671)</td></tr><tr><td align="left" valign="top">Overall quality score</td><td align="left" valign="top" colspan="2">0.272 (0.193&#x2010;0.342)</td><td align="left" valign="top" colspan="2">0.529 (0.418&#x2010;0.609)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>ICC: intraclass correlation coefficient.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Quality Comparisons</title><p>Overall quality ratings were compared between CA-generated and clinician-generated responses. Descriptive statistics were calculated using response-level scores averaged across 3 clinician raters, whereas mean differences and CIs were estimated from linear mixed effects models. Across 288 unique case-question responses, CA responses received higher overall quality scores than clinician responses (<xref ref-type="table" rid="table3">Table 3</xref>). The mean overall quality score was 4.37 (SD 0.57) for CA responses and 3.58 (SD 0.90) for clinician responses. The estimated mean difference was 0.782 (95% CI 0.692&#x2010;0.872; <italic>P</italic>&#x003C;.001). Median scores were also higher for CA responses (4.5, IQR 4.0&#x2010;4.8) than for clinician responses (3.8, IQR 3.0&#x2010;4.2), indicating a consistent shift in central tendency. Variability differed between response types, with lower dispersion observed for CA scores (SD 0.57) relative to clinician scores (SD 0.90); this pattern suggests lower between-response dispersion in CA ratings.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Overall and dimension-specific quality ratings for conversational agent (CA) and clinician responses. CA and clinician responses were rated on 6 quality dimensions (1-5), with the overall quality score defined as the mean of the 6 items.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Outcome</td><td align="left" valign="bottom">CA<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup> mean (SD)</td><td align="left" valign="bottom">CA median (IQR)</td><td align="left" valign="bottom">Clinician mean (SD)</td><td align="left" valign="bottom">Clinician median (IQR)</td><td align="left" valign="bottom">Mean difference (95% CI)<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Overall quality</td><td align="left" valign="top">4.37 (0.57)</td><td align="left" valign="top">4.5 (4.0&#x2010;4.8)</td><td align="left" valign="top">3.58 (0.90)</td><td align="left" valign="top">3.8 (3.0&#x2010;4.2)</td><td align="left" valign="top">0.782 (0.692&#x2010;0.872)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Clinical accuracy</td><td align="left" valign="top">4.40 (0.69)</td><td align="left" valign="top">4.0 (4.0&#x2010;5.0)</td><td align="left" valign="top">3.84 (0.95)</td><td align="left" valign="top">4.0 (3.0&#x2010;4.0)</td><td align="left" valign="top">0.562 (0.463&#x2010;0.662)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Guideline adherence</td><td align="left" valign="top">4.31 (0.73)</td><td align="left" valign="top">4.0 (4.0&#x2010;5.0)</td><td align="left" valign="top">3.82 (0.92)</td><td align="left" valign="top">4.0 (3.0&#x2010;4.0)</td><td align="left" valign="top">0.495 (0.394&#x2010;0.597)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Actionability</td><td align="left" valign="top">4.42 (0.71)</td><td align="left" valign="top">5.0 (4.0&#x2010;5.0)</td><td align="left" valign="top">3.43 (1.13)</td><td align="left" valign="top">4.0 (3.0&#x2010;4.0)</td><td align="left" valign="top">0.992 (0.877&#x2010;1.106)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Personalization</td><td align="left" valign="top">4.25 (0.75)</td><td align="left" valign="top">4.0 (4.0&#x2010;5.0)</td><td align="left" valign="top">3.38 (1.11)</td><td align="left" valign="top">4.0 (3.0&#x2010;4.0)</td><td align="left" valign="top">0.867 (0.749&#x2010;0.985)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Clarity</td><td align="left" valign="top">4.44 (0.65)</td><td align="left" valign="top">4.0 (4.0&#x2010;5.0)</td><td align="left" valign="top">3.73 (1.04)</td><td align="left" valign="top">4.0 (3.0&#x2010;5.0)</td><td align="left" valign="top">0.713 (0.611&#x2010;0.815)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Empathy</td><td align="left" valign="top">4.37 (0.65)</td><td align="left" valign="top">4.0 (4.0&#x2010;5.0)</td><td align="left" valign="top">3.30 (1.19)</td><td align="left" valign="top">3.0 (3.0&#x2010;4.0)</td><td align="left" valign="top">1.062 (0.948&#x2010;1.177)</td><td align="left" valign="top">&#x003C;.001</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>CA: conversational agent.</p></fn><fn id="table3fn2"><p><sup>b</sup>Mean differences (CA &#x2212; clinician) were estimated using linear mixed effects models with random intercepts for case and rater.</p></fn></table-wrap-foot></table-wrap><p>To examine whether this overall pattern was consistent across clinician-authored responses, <xref ref-type="fig" rid="figure2">Figure 2</xref> presents overall quality score distributions stratified by the clinician who authored the human responses. In each stratum, the distribution of CA scores was shifted upward relative to clinician-authored responses. CA scores restricted to the same case subsets (&#x201C;matched&#x201D;) were comparable to the overall CA distribution, indicating that the observed difference was not confined to particular case allocations.</p><p>In sensitivity analyses using response-level matched case&#x2013;question pairs (n=144), the distribution of paired differences deviated from normality (Shapiro-Wilk <italic>P</italic>=.001). The Wilcoxon signed-rank test confirmed significantly higher scores for CA responses (W=479; <italic>P</italic>&#x003C;.001), with a median difference of 0.82 points. The estimated rank-biserial correlation (r&#x2248;0.95) indicated a large effect size.</p><p>Response length differed markedly between CA and clinician responses. The mean word count was 211.4 (SD 54.8; 95% CI 202.4&#x2010;220.4) for CA responses compared with 72.9 (SD 68.8; 95% CI 61.6&#x2010;84.2) for clinician responses, representing nearly a 3-fold difference in verbosity. Substantial variability in response length was also observed across individual clinicians. Mean word count ranged from 39.5 (SD 15.4; 95% CI 33.0&#x2010;46.0) to 191.3 (SD 89.2; 95% CI 153.6&#x2010;228.9). Interclinician variability was statistically significant (Kruskal-Wallis H=65.83 [<xref ref-type="bibr" rid="ref66">66</xref>]; <italic>P</italic>&#x003C;.001; &#x03B5;&#x00B2;=0.44) and remained significant after excluding the highest-verbosity clinician (H=20.70; <italic>P</italic>&#x003C;.001).</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Overall quality score distributions for clinician-authored and conversational agent (CA) responses. Each panel corresponds to one clinician (A-F) and compares clinician-authored responses, CA responses matched to the same cases and questions, and CA responses across all evaluated case-question pairs. Overall quality scores were calculated as the mean of 6 evaluation dimensions. &#x201C;M&#x201D; denotes the median; boxes represent the IQR, whiskers indicate the observed range, and points represent individual response-level scores. CA: conversational agent.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e98519_fig02.png"/></fig><p>Despite these marked differences in response length, word count was not significantly associated with overall quality ratings in mixed effects models (<xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref>). Neither the main effect of word count nor its interaction with responder type was statistically significant for the overall score (all <italic>P</italic>&#x003E;.14). Among individual quality dimensions, only empathy demonstrated a significant interaction between word count and responder type (<italic>P</italic>=.003). In clinician-authored responses, word count was negatively associated with empathy ratings (&#x03B2;=&#x2212;0.00235 per word; <italic>P</italic>&#x003C;.001), whereas no significant association was observed for CA responses (&#x03B2;=0.00107; <italic>P</italic>=.22). Overall, we did not find evidence that word count explained the observed differences in quality between CA and clinician responses.</p><p>Dimension-specific mixed effects analyses identified statistically significant differences between CA and clinician responses across all 6 quality dimensions (all <italic>P</italic>&#x003C;.001). Estimated mean differences ranged from 0.495 to 1.062 (<xref ref-type="table" rid="table3">Table 3</xref>). The largest differences were observed for empathy (1.062; 95% CI 0.948&#x2010;1.177) and actionability (0.992; 95% CI 0.877&#x2010;1.106), whereas smaller differences were observed for clinical accuracy (0.562; 95% CI 0.463&#x2010;0.662) and guideline adherence (0.495; 95% CI 0.394&#x2010;0.597). All estimated mean differences were positive.</p><p>Across the 6 predefined content domains, mixed effects models likewise indicated statistically significant differences in overall quality between CA and clinician responses (all <italic>P</italic>&#x2264;.025; <xref ref-type="supplementary-material" rid="app7">Multimedia Appendix 7</xref>). Estimated mean differences ranged from 0.419 to 1.013. The largest estimated difference was observed in Domain A (blood glucose interpretation and fluctuation analysis; mean difference 1.013, 95% CI 0.762&#x2010;1.265), followed by Domain C (medication and treatment guidance; 0.912, 95% CI 0.657&#x2010;1.168).</p><p>Notably, significant differences were also observed in domain 4, which comprised emotional and psychological concerns (eg, stress-related glycemic dysregulation, fear of hypoglycemia, and feelings of frustration or self-doubt; estimated mean difference 0.721, 95% CI 0.507&#x2010;0.935; <italic>P</italic>&#x003C;.001). In this psychosocial domain, CA responses were consistently rated higher across quality dimensions, including empathy and actionability, indicating that performance differences were not confined to technical CGM interpretation but extended to emotionally sensitive scenarios.</p><p>The smallest estimated differences were observed in domain 5 (long-term goals and motivation; 0.419, 95% CI 0.051&#x2010;0.786) and domain 6 (technical issues and device use; 0.493, 95% CI 0.065&#x2010;0.921). <xref ref-type="fig" rid="figure3">Figure 3</xref> visualizes the domain- and dimension-specific mean quality ratings, illustrating a consistent pattern of higher scores for CA responses across all 6 content domains and quality dimensions. Full descriptive statistics (mean, SD) are provided in <xref ref-type="supplementary-material" rid="app8">Multimedia Appendix 8</xref>.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Mean quality ratings by clinical domain and dimension. Heatmaps compare conversational agent (CA)&#x2013;generated and clinician-generated responses across 6 clinical domains and 6 quality dimensions. Values represent domain-level means of response-level scores averaged across 3 clinician ratings, and color intensity indicates the mean rating on a 1-5 scale. CA: conversational agent.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e98519_fig03.png"/></fig></sec><sec id="s3-3"><title>Post Hoc Retrieval Audit</title><p>A post hoc retrieval audit (<xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>) was conducted on 24 domain-stratified CA responses, including 4 case-question pairs from each of the 6 predefined question domains. As described in the Methods, the audit sample was selected to maximize domain and case-question coverage rather than to provide a statistically representative estimate of retrieval-response alignment across all CA responses. Across the audited examples, retrieved segments were directly aligned with the corresponding CA response in 8 out of 24 examples (33.3%), broadly aligned in 7 out of 24 (29.2%), weakly aligned in 5 out of 24 (20.8%), and showed no clear alignment in 4 out of 24 (16.7%). Thus, 15 out of 24 audited responses (62.5%) showed direct or broad retrieval-response alignment, while 20 out of 24 (83.3%) showed at least weak topical or contextual relevance. Conversely, 9 out of 24 (37.5%) audited examples showed only weak or no clear alignment, indicating that visible retrieval-response alignment was incomplete and varied across domains.</p><p>The mean FAISS L2 distance across audited items was 0.988. Lower L2 distances were directionally associated with stronger manual alignment categories, although the retrieval score was not treated as a standalone relevance measure. The mean L2 distance was 0.855 for directly aligned examples, 0.848 for broadly aligned examples, 1.159 for weakly aligned examples, and 1.288 for examples with no clear alignment. This pattern suggests that embedding-space proximity provided a useful quantitative retrieval trace, but manual review remained necessary to determine clinical relevance and visible response grounding.</p><p>Domain-level patterns showed the strongest retrieval-response alignment in blood glucose interpretation and fluctuation analysis, where all 4 audited examples were directly aligned. Medication and treatment guidance and emotional and psychological concerns also showed relatively strong alignment, with direct or broad alignment in 3 of 4 examples in each domain. Long-term goals and motivation showed at least weak alignment in all 4 examples, although only 2 of 4 were directly or broadly aligned. Alignment was weakest for technical issues and device use, where only 1 of 4 examples showed direct or broad alignment and 2 of 4 showed no clear alignment.</p></sec><sec id="s3-4"><title>Safety and Source Identification</title><p>Beyond quantitative quality comparisons, we examined raters&#x2019; ability to identify the source of each response and the distribution of safety flags.</p><p>Across 864 ratings, clinicians correctly identified the source in 697 (80.7%) cases, misclassified 100 (11.6%), and selected &#x201C;not sure&#x201D; in 67 (7.8%). Restricting the analysis to definitive judgments (n=797 rating-level classifications), overall identification accuracy was 87.5% (one-sided exact binomial test vs 50% chance, <italic>P</italic>&#x003C;.001), indicating that responses were generally distinguishable from one another. These findings indicate that, although the response source was masked during the rating task, CA and clinician responses were often distinguishable in practice.</p><p>Identification performance varied across raters. 5 clinicians demonstrated high discrimination accuracy (82.3%&#x2010;100.0% among definitive judgments), whereas one clinician did not perform above chance level (50.0%; <italic>P</italic>=.54), suggesting substantial interrater heterogeneity. Rater-specific distributions of overall quality scores, stratified by perceived source (clinician vs CA), are shown in <xref ref-type="supplementary-material" rid="app9">Multimedia Appendix 9</xref>. Descriptively, responses perceived as CA tended to receive higher median quality scores for most raters, although this pattern was not uniform and some raters showed similar or lower scores for responses perceived as CA. This analysis was exploratory and was not interpreted as a causal test of source-label bias because perceived source, true source, response style, and response structure were closely entangled.</p><p>Safety flag distributions were comparable between sources. Among clinician responses (n=432), 387 (89.6%) were rated as Level 0, 42 (9.7%) as Level 1, and 3 (0.7%) as Level 2. Among CA responses (n=432), corresponding proportions were 389 (90.1%), 40 (9.3%), and 3 (0.7%), respectively. However, qualitative comments were available for only 23 flagged ratings, as written explanations were optional, and therefore not all safety flags were accompanied by narrative justification.</p><p>Among the 23 available comments, most (n=15, 65.2%) concerned glucagon-like peptide-1 (GLP-1) eligibility under NHS BMI criteria, with most of these applying to CA responses (13 CA vs 2 clinicians). Three (13.0%) comments related to medication administration, specifically acarbose dosing (2 CA vs 1 clinician), and 2 (8.7%) comments noted that the response did not directly address the question. The remaining comments (n=3, all CA) referred to behavioral feasibility concerns, including the perceived burden of the proposed action plan and the appropriateness of specific dietary substitution advice. These comment-level data should be interpreted cautiously because written remarks were optional, but they suggest that similar flag frequencies do not necessarily imply identical patterns of concern across response sources.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>In this source-masked multirater evaluation of vignette-based CGM-informed scenarios, the retrieval-grounded CA received higher mean structured quality ratings than clinician-authored responses across predefined domains. The overall mean difference was approximately 0.8 points on a 5-point scale, with the largest differences observed in empathy (mean difference 1.06) and actionability (0.99). These findings suggest that, under standardized vignette-based conditions, a scaffolded CA output pipeline incorporating retrieval as part of its architecture can produce written responses that specialists judge favorably in both relational and action-oriented dimensions. However, the comparison should be interpreted as a controlled evaluation of written outputs rather than as evidence that the CA is intrinsically superior to clinicians in real-world diabetes counseling. To our knowledge, this represents one of the first source-masked comparative evaluations of retrieval-grounded LLM-generated and clinician-authored responses in structured CGM counseling contexts.</p><p>Performance differences were not uniform across domains and appeared to vary according to task structure. Differences were most pronounced in data-intensive glucose interpretation and action-oriented explanation and attenuated in domains centered on long-term motivational support and device troubleshooting. Notably, significant differences were also observed in psychosocial scenarios, indicating that comparative ratings were not limited to quantitative glucose analysis but extended to contexts involving emotional distress, fear of hypoglycemia, and diabetes-related frustration. This gradient suggests that relative performance may depend on the cognitive structure of the task. The scaffolded CA pipeline evaluated here appeared particularly well suited to synthesizing numerical CGM metrics, vignette context, structured prompting, and guideline-oriented reference material into structured explanations and action plans [<xref ref-type="bibr" rid="ref25">25</xref>]. In contrast, domains requiring nuanced behavioral coaching or experiential clinical judgment may depend more heavily on individualized framing [<xref ref-type="bibr" rid="ref67">67</xref>,<xref ref-type="bibr" rid="ref68">68</xref>].</p><p>The post hoc retrieval audit provides additional context for interpreting the contribution of the RAG component. The audit suggested that retrieval grounding was partial and domain-dependent rather than uniform across all CA outputs. Retrieval-response alignment was strongest for blood glucose interpretation and fluctuation analysis, where all audited examples were directly aligned, and was also relatively strong for medication and treatment guidance and emotional and psychological concerns. This pattern suggests that the curated corpus was most useful when questions could be supported by guideline-based or educational material directly relevant to CGM interpretation, treatment-context explanation, or psychosocial support. In contrast, technical issues and device use showed weaker visible retrieval-response alignment, suggesting that future versions may require more targeted device-specific retrieval materials, such as manufacturer guidance, sensor troubleshooting documentation, and local clinical protocols. Therefore, the RAG component should be interpreted as providing retrieval traceability and visible response alignment in a subset of cases, rather than as uniformly determining CA outputs or explaining the observed quality differences.</p></sec><sec id="s4-2"><title>Comparison With Prior Work</title><p>These findings extend prior CGM-focused evaluations of LLM-generated summaries, which have typically assessed model outputs in isolation and emphasized feasibility, accuracy, or clinical acceptability [<xref ref-type="bibr" rid="ref31">31</xref>]. By using a balanced design with source-masked ratings, this study enables direct comparison with clinician-authored responses across predefined technical and relational dimensions. Notably, the largest differences were observed in actionability and empathy, rather than being confined solely to glycemic interpretation, suggesting that structured explanatory and relational components may be particularly sensitive to comparative evaluation.</p><p>Related work in other wearable-data domains has explored the use of AI systems to interpret structured physiological data streams, such as heart rate, physical activity, or sleep measures generated by consumer wearables [<xref ref-type="bibr" rid="ref69">69</xref>-<xref ref-type="bibr" rid="ref71">71</xref>]. These studies have similarly focused on automated summarization, health insight generation, or behavioral coaching based on wearable-derived data [<xref ref-type="bibr" rid="ref70">70</xref>]. However, most have evaluated system outputs in isolation or against guideline-based expectations rather than through source-masked comparison with clinician-authored responses. This study therefore contributes additional evidence by examining how LLM-generated explanations compare directly with clinician communication in a structured evaluation setting.</p><p>The evaluation was confined to structured, vignette-based scenarios involving common CGM-related questions, concept clarification, and general diabetes self-management guidance. It did not assess complex therapeutic decision-making, individualized prescribing adjustments, or real-time risk management. Accordingly, these findings should be interpreted within the scope of clearly bounded explanatory tasks, rather than extended to higher-stakes clinical reasoning contexts.</p><p>Beyond CGM-specific evaluations, these findings align with broader diabetes care literature, suggesting cautious optimism regarding AI-supported communication tools. Reported areas of potential utility include patient education, explanation of structured health data, and information synthesis, whereas greater caution is typically expressed when systems are used to support individualized treatment decisions or medication adjustments [<xref ref-type="bibr" rid="ref68">68</xref>,<xref ref-type="bibr" rid="ref72">72</xref>]. Within this context, scaffolded CAs incorporating retrieval may be best viewed as adjunct tools for patient-facing explanation in clearly defined, guideline-constrained use cases, with clinician oversight maintained for interpretation and final clinical framing [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref67">67</xref>-<xref ref-type="bibr" rid="ref72">72</xref>].</p><p>Although workflow outcomes were not directly evaluated, the observed performance in standardized CGM explanation tasks suggests potential relevance to clinical workflow. In routine practice, summarizing CGM trends, clarifying commonly used thresholds, and addressing frequent educational queries can consume substantial consultation time [<xref ref-type="bibr" rid="ref46">46</xref>]. If deployed within appropriately bounded tasks and aligned to local guidance and policy constraints, scaffolded CAs incorporating retrieval may help support these informational components of care. Any such use should be framed as supportive rather than substitutive, and prospective studies are needed to quantify effects on clinician workload, consultation flow, and patient outcomes.</p></sec><sec id="s4-3"><title>Limitations</title><p>Several factors limit how these findings should be interpreted and generalized. First, although the study incorporated safeguards against direct self-rating and single-rater effects, the clinician panel was not fully independent of the comparator-generation process because the same group of specialists authored comparator responses in Phase 1 and rated anonymized responses in Phase 2. These safeguards reduce the risk of direct self-rating and idiosyncratic single-rater judgments, but they do not remove the possibility that clinicians brought expectations from their own authoring experience into the rating phase. Interrater agreement was also modest, with single-rater reliability ranging from poor to fair across most dimensions (ICC(1,2) range: 0.087&#x2010;0.324). This likely reflects the difficulty of rating complex patient-facing clinical narratives, where expert judgments may vary according to clinical training background, subspecialty experience, local practice norms, communication style, and individual interpretation of rating dimensions such as guideline adherence, actionability, and empathy [<xref ref-type="bibr" rid="ref73">73</xref>-<xref ref-type="bibr" rid="ref75">75</xref>]. Reliability improved when ratings were averaged across 3 clinicians, supporting the use of aggregated multirater scores for primary interpretation. However, dimension-specific findings, particularly those based on modest absolute differences, should be interpreted cautiously because the rating instrument and sample size may not support strong inferences at the individual-dimension level.</p><p>Second, the case set was small and included only 3 type 2 diabetes cases. This limits generalizability to routine diabetes care, where type 2 diabetes accounts for the majority of diagnosed diabetes [<xref ref-type="bibr" rid="ref76">76</xref>]. However, this imbalance also reflects a broader data-availability constraint, as CGM use and publicly available CGM datasets remain disproportionately concentrated in type 1 diabetes and insulin-treated populations [<xref ref-type="bibr" rid="ref77">77</xref>,<xref ref-type="bibr" rid="ref78">78</xref>]. More broadly, the study had limited coverage of medication-policy-sensitive scenarios, including questions involving treatment escalation, eligibility criteria, cardiometabolic risk management, and locally specific prescribing rules. At the aggregate level, safety flag frequencies were similar between CA and clinician responses, and major safety concerns were rare. However, optional written safety comments were more often attached to CA responses (n=20, 87%) and most frequently concerned GLP-1 eligibility under NHS BMI criteria [<xref ref-type="bibr" rid="ref74">74</xref>]. This should be interpreted primarily as a local prescribing and policy-alignment issue rather than as a general CGM interpretation error. It also suggests that similar aggregate safety flag frequencies should not be interpreted as identical safety profiles across response sources, diabetes types, or treatment contexts. Larger studies with more diverse type 1 and type 2 diabetes cases, including different treatment regimens, BMI profiles, medication eligibility scenarios, and local prescribing policies, are needed before extending these findings to routine diabetes care.</p><p>Third, the comparison should be interpreted in light of imperfect source masking, response recognizability, and potential development-stage familiarity bias. Although raters were not informed of the response source and all responses were anonymized, source-identification accuracy was high, indicating that the evaluation was source-masked by design but imperfectly blinded in practice. This recognizability likely reflected deliberate features of the CA configuration, including structured prompting, safety-bounded generation, plain-language communication guidance, and a low temperature setting intended to reduce stochastic variation and improve reproducibility [<xref ref-type="bibr" rid="ref79">79</xref>]. These choices were intended to derisk patient-facing communication, but they may also have increased the consistency and recognizability of CA outputs.</p><p>In addition, one senior diabetologist provided formative feedback during vignette and question-bank development and later participated as one of the clinician raters. This creates potential circularity or familiarity bias, particularly regarding expectations about appropriate response structure, tone, or clinical framing. This risk was reduced because the pilot profile reviewed during development was not included in the final case library, development-stage response excerpts were not rated as formal outputs, and the final evaluation used a source-masked multirater design involving 6 clinicians. However, these safeguards do not fully remove the possibility that response style, source recognizability, or development-stage familiarity influenced ratings, particularly for empathy and personalization.</p><p>The perceived-source analysis suggested rater-specific heterogeneity: responses perceived as CA often received higher scores, but this pattern was not consistent across all raters, indicating that perceptual bias, if present, was unlikely to operate uniformly. Personalization ratings may also have been affected by the CA&#x2019;s more explicit use of case-specific CGM and vignette details, whereas clinicians may have conveyed individualized judgment more concisely or implicitly. Although word-count-adjusted analyses did not indicate that response length explained the overall quality difference, response style and recognizability may still have influenced how raters applied relational criteria. Higher empathy ratings for CA responses should therefore be interpreted as higher perceived empathy of the written outputs under this evaluation design, rather than as evidence of intrinsic empathic capacity or superiority over clinician communication in real-world consultations. Future studies should separate vignette development, system refinement, and outcome rating across independent clinical panels where feasible.</p><p>A related limitation is the asymmetry between the scaffolded CA configuration and the clinician-authored response condition. The clinician contributors were senior diabetes specialists with substantial clinical experience, tacit guideline knowledge, and real-world communication expertise. Their responses, therefore, should not be interpreted as reflecting a lack of clinical knowledge or capability, but as usual, expert-authored written communication under the constraints of a vignette-based task. By contrast, the CA was explicitly configured to produce standardized written explanations using structured prompting, communication and safety instructions, RAG-retrieved reference material, and low-temperature generation. These design features may have improved consistency, completeness, structure, and alignment with the study rating rubric, particularly for clarity, actionability, and empathy. Accordingly, the observed quality differences should be interpreted as differences between a scaffolded retrieval-grounded CA output pipeline and usual clinician-authored written responses under controlled vignette conditions, rather than as evidence that the CA is intrinsically superior to clinicians in clinical communication or real-world diabetes counseling. In real clinical practice, clinicians provide interactive clarification, longitudinal knowledge of the patient, contextual judgment, and responsibility for treatment decisions, which were not fully represented in this single-turn written evaluation. Future studies could examine complementary comparison conditions, such as clinician-authored responses supported by structured templates, shared reference materials, or CA-assisted drafting workflows, to distinguish the effect of response scaffolding from the effect of model generation itself.</p><p>Fourth, the findings reflect a single retrieval-grounded system and one model configuration. Generation parameters, including the low temperature setting and API access timing, are reported in the Methods to improve reproducibility [<xref ref-type="bibr" rid="ref79">79</xref>]; however, outputs may differ under alternative models, API snapshots, retrieval resources, chunking strategies, prompts, or generation settings. The post hoc retrieval audit also showed that RAG contribution was partial and domain-dependent rather than uniformly visible across all outputs. Some responses were directly or broadly aligned with retrieved material, whereas others showed only weak or no clear visible contribution from the top-ranked retrieved segment. Importantly, this audit was designed to assess retrieval traceability and visible response alignment, not to establish a causal effect of retrieval on response quality. Without an ablation comparison, this study cannot determine how much of the observed performance reflects retrieval grounding rather than the base model&#x2019;s parametric knowledge, structured prompting, CGM summary inputs, vignette context, or safety-oriented response instructions. Accordingly, &#x201C;retrieval-grounded&#x201D; should be understood here as a description of the system architecture and input pipeline rather than as evidence that retrieval was the primary driver of the observed quality ratings. The weaker alignment observed for technical issues and device use likely reflects the scope of the current corpus, which included limited sensor-use and device-troubleshooting material. Future work should include full retrieval logging, ablation comparisons, independent relevance assessment, and expanded device-specific and local policy-specific retrieval corpora.</p><p>Finally, the study used written, vignette-based scenarios rather than interactive real-world consultations. Although vignettes were derived from real CGM data and reviewed for clinical plausibility, they necessarily simplified real-world encounters and did not encompass rare presentations, highly complex comorbidities, diverse treatment pathways, or dynamic patient-clinician interaction. The findings are therefore most applicable to structured CGM interpretation and patient-facing explanation tasks under controlled conditions. Future studies should evaluate scaffolded retrieval-grounded systems prospectively in clinical workflows, such as preconsultation preparation or supervised patient education, and assess patient understanding, consultation efficiency, clinician workload, safety escalation, and local governance requirements.</p></sec><sec id="s4-4"><title>Conclusions</title><p>Taken together, these findings support a potential role for scaffolded LLM systems incorporating retrieval as part of their architecture as adjunct tools for structured CGM interpretation and patient-facing explanation of common questions arising in routine diabetes care under controlled, source-masked vignette-based conditions. In practical terms, their most appropriate role appears to be explanatory and educational rather than therapeutic: they may help patients understand glucose patterns, clarify common concerns, and prepare for discussion with their diabetes care team, including in relation to emotionally sensitive issues. However, this should not be interpreted as support for individualized therapeutic decision-making. In particular, such systems are not established here as appropriate for recommending medication initiation, dose adjustment, regimen change, or other personalized treatment decisions, which should remain clinician-led. Nor do these findings establish equivalence to clinician communication in real-world consultations, where ongoing therapeutic relationships, contextual judgment, and dynamic interaction remain central. The findings should also be interpreted in light of the imperfect source masking and the asymmetry between scaffolded CA outputs and unassisted clinician-authored responses. Further prospective evaluation in interactive clinical settings is needed to define the appropriate scope, safeguards, and implementation of such systems in practice.</p></sec></sec></body><back><ack><p>The authors would like to thank the clinician contributors who contributed their time and expertise to this study. We also sincerely thank the 6 senior diabetes clinicians for their valuable support, expert input, and careful contribution to this work. Their clinical expertise, thoughtful feedback, and engagement in the evaluation process were essential to the development and refinement of this study. We are also grateful to colleagues and collaborators who provided helpful discussions and support during the design, implementation, and preparation of this research.</p><p>Use of Generative AI</p><p>During manuscript revision, the authors used ChatGPT to support language editing, wording refinement, and formatting consistency, including assistance with reference style checking. Generative AI was not used to generate study data, conduct statistical analyses, assign clinical ratings, adjudicate safety outcomes, or determine the study conclusions. All AI-assisted text and references were reviewed, edited, and verified by the authors. The authors take full responsibility for the accuracy, integrity, and final content of the manuscript.</p></ack><notes><sec><title>Funding</title><p>This study was supported by internal project funding from the UCL Institute of Health Informatics, the University College London Hospitals Biomedical Research Center, and the Moorfields Biomedical Research Center. The funders had no role in the study design; data collection, analysis, or interpretation; manuscript writing; or the decision to submit the article for publication.</p></sec><sec><title>Data Availability</title><p>The ShanghaiT1DM and ShanghaiT2DM datasets are publicly available via Figshare [<xref ref-type="bibr" rid="ref36">36</xref>]. The OhioT1DM dataset is publicly available via Kaggle [<xref ref-type="bibr" rid="ref37">37</xref>].</p><p>The curated retrieval corpus used for the RAG system, derived exclusively from publicly available clinical guidelines and educational materials, is available on GitHub [<xref ref-type="bibr" rid="ref80">80</xref>]. No new patient data were generated in this study.</p><p>The CA implementation, including the RAG pipeline and analysis scripts used in this study.</p></sec></notes><fn-group><fn fn-type="con"><p>ZG and KL contributed to the conception and design of the study. ZG led the development and deployment of the system. ZG and KL coordinated participant recruitment and study facilitation. EK reviewed the question bank, assessed the patient profiles, and provided targeted clinical feedback during early-stage system development. EK, AV, IA, CA, and JH provided clinical input and expert feedback on the study design, system development, and manuscript. ZG wrote the first draft of the manuscript. All authors contributed to the review and revision of the manuscript. All authors read and approved the final version of the manuscript.</p></fn><fn fn-type="conflict"><p>The authors declare no financial competing interests. Several coauthors were involved as source-masked raters and/or contributors of comparator responses. To minimize potential bias, raters were not informed of response source and did not assess their own responses. Their involvement in manuscript review and revision was limited to interpretation, clinical contextualization, and critical review of the manuscript.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ADA</term><def><p>American Diabetes Association</p></def></def-item><def-item><term id="abb2">AGP</term><def><p>ambulatory glucose profile</p></def></def-item><def-item><term id="abb3">CA</term><def><p>conversational agent</p></def></def-item><def-item><term id="abb4">CGM</term><def><p>continuous glucose monitoring</p></def></def-item><def-item><term id="abb5">DECIDE-AI</term><def><p>Developmental and Exploratory Clinical Investigations of Decision Support Systems Driven by AI</p></def></def-item><def-item><term id="abb6">FAISS</term><def><p>Facebook AI Similarity Search</p></def></def-item><def-item><term id="abb7">GLP-1</term><def><p>glucagon-like peptide-1</p></def></def-item><def-item><term id="abb8">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb9">IDF</term><def><p>International Diabetes Federation</p></def></def-item><def-item><term id="abb10">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb11">NHS</term><def><p>National Health Service</p></def></def-item><def-item><term id="abb12">RAG</term><def><p>retrieval-augmented generation</p></def></def-item><def-item><term id="abb13">TAR</term><def><p>time above range</p></def></def-item><def-item><term id="abb14">TBR</term><def><p>time below range</p></def></def-item><def-item><term id="abb15">TIR</term><def><p>time in range</p></def></def-item><def-item><term id="abb16">UCLH</term><def><p>University College London Hospitals</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="web"><article-title>Diabetes</article-title><source>National Health Service</source><year>2025</year><access-date>2026-07-15</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.nhs.uk/conditions/diabetes/">https://www.nhs.uk/conditions/diabetes/</ext-link></comment></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="web"><article-title>Diabetes facts &#x0026; figures</article-title><source>International Diabetes Federation</source><year>2025</year><access-date>2026-07-15</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://idf.org/about-diabetes/diabetes-facts-figures/">https://idf.org/about-diabetes/diabetes-facts-figures/</ext-link></comment></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bian</surname><given-names>H</given-names> </name><name name-style="western"><surname>Zhan</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ni</surname><given-names>L</given-names> </name><name name-style="western"><surname>Huo</surname><given-names>L</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>J</given-names> </name></person-group><article-title>Digital management of diabetes global research trends: a bibliometric study</article-title><source>Front Med (Lausanne)</source><year>2025</year><volume>12</volume><fpage>1620307</fpage><pub-id pub-id-type="doi">10.3389/fmed.2025.1620307</pub-id><pub-id pub-id-type="medline">41164162</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>S</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>S</given-names> </name><etal/></person-group><article-title>A community-codesigned LLM-powered chatbot for primary care: a randomized controlled trial</article-title><source>Nat Health</source><year>2026</year><volume>1</volume><issue>2</issue><fpage>238</fpage><lpage>250</lpage><pub-id pub-id-type="doi">10.1038/s44360-025-00021-w</pub-id><pub-id pub-id-type="medline">41659358</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Klupa</surname><given-names>T</given-names> </name><name name-style="western"><surname>Czupryniak</surname><given-names>L</given-names> </name><name name-style="western"><surname>Dzida</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Expanding the role of continuous glucose monitoring in modern diabetes care beyond type 1 disease</article-title><source>Diabetes Ther</source><year>2023</year><month>08</month><volume>14</volume><issue>8</issue><fpage>1241</fpage><lpage>1266</lpage><pub-id pub-id-type="doi">10.1007/s13300-023-01431-3</pub-id><pub-id pub-id-type="medline">37322319</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Bergenstal</surname><given-names>RM</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Hirsch</surname><given-names>IB</given-names> </name></person-group><article-title>Understanding continuous glucose monitoring data</article-title><source>Role of Continuous Glucose Monitoring in Diabetes Treatment</source><year>2018</year><publisher-name>American Diabetes Association</publisher-name><fpage>20</fpage><lpage>23</lpage><pub-id pub-id-type="doi">10.2337/db20181-20</pub-id><pub-id pub-id-type="medline">34251769</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Uduku</surname><given-names>C</given-names> </name><name name-style="western"><surname>Li</surname><given-names>K</given-names> </name><name name-style="western"><surname>Herrero</surname><given-names>P</given-names> </name><name name-style="western"><surname>Oliver</surname><given-names>N</given-names> </name><name name-style="western"><surname>Georgiou</surname><given-names>P</given-names> </name></person-group><article-title>Enhancing self-management in type 1 diabetes with wearables and deep learning</article-title><source>NPJ Digit Med</source><year>2022</year><month>06</month><day>27</day><volume>5</volume><issue>1</issue><fpage>78</fpage><pub-id pub-id-type="doi">10.1038/s41746-022-00626-5</pub-id><pub-id pub-id-type="medline">35760819</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bendixen</surname><given-names>BE</given-names> </name><name name-style="western"><surname>Wilhelmsen-Langeland</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lomborg</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Intermittent use of continuous glucose monitoring in type 2 diabetes is preferred: a qualitative study of patients&#x2019; experiences</article-title><source>Sci Diabetes Self Manag Care</source><year>2025</year><month>06</month><volume>51</volume><issue>3</issue><fpage>323</fpage><lpage>332</lpage><pub-id pub-id-type="doi">10.1177/26350106251326517</pub-id><pub-id pub-id-type="medline">40116013</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>ElSayed</surname><given-names>NA</given-names> </name><name name-style="western"><surname>Aleppo</surname><given-names>G</given-names> </name><name name-style="western"><surname>Aroda</surname><given-names>VR</given-names> </name><etal/></person-group><article-title>9. Pharmacologic approaches to glycemic treatment: standards of care in diabetes-2023</article-title><source>Diabetes Care</source><year>2023</year><month>01</month><day>1</day><volume>46</volume><issue>Suppl 1</issue><fpage>S140</fpage><lpage>S157</lpage><pub-id pub-id-type="doi">10.2337/dc23-S009</pub-id><pub-id pub-id-type="medline">36507650</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Martens</surname><given-names>TW</given-names> </name><name name-style="western"><surname>Simonson</surname><given-names>GD</given-names> </name><name name-style="western"><surname>Bergenstal</surname><given-names>RM</given-names> </name></person-group><article-title>Using continuous glucose monitoring data in daily clinical practice</article-title><source>Cleve Clin J Med</source><year>2024</year><month>10</month><day>1</day><volume>91</volume><issue>10</issue><fpage>611</fpage><lpage>620</lpage><pub-id pub-id-type="doi">10.3949/ccjm.91a.23090</pub-id><pub-id pub-id-type="medline">39353661</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mayberry</surname><given-names>LS</given-names> </name><name name-style="western"><surname>Guy</surname><given-names>C</given-names> </name><name name-style="western"><surname>Hendrickson</surname><given-names>CD</given-names> </name><name name-style="western"><surname>McCoy</surname><given-names>AB</given-names> </name><name name-style="western"><surname>Elasy</surname><given-names>T</given-names> </name></person-group><article-title>Rates and correlates of uptake of continuous glucose monitors among adults with type 2 diabetes in primary care and endocrinology settings</article-title><source>J Gen Intern Med</source><year>2023</year><month>08</month><volume>38</volume><issue>11</issue><fpage>2546</fpage><lpage>2552</lpage><pub-id pub-id-type="doi">10.1007/s11606-023-08222-3</pub-id><pub-id pub-id-type="medline">37254011</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="web"><article-title>Dexcom clarity reports overview</article-title><source>Dexcom</source><access-date>2026-04-02</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://provider.dexcom.com/education-research/cgm-education-use/product-information/dexcom-clarity-reports-overview">https://provider.dexcom.com/education-research/cgm-education-use/product-information/dexcom-clarity-reports-overview</ext-link></comment></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="web"><person-group person-group-type="author"><collab>Abbott</collab></person-group><article-title>FreeStyle libre software reports tour</article-title><source>LibreView</source><year>2024</year><access-date>2026-04-02</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.libreview.com/files/documents/en-GB/FSReportTour_2024-08-19.pdf">https://www.libreview.com/files/documents/en-GB/FSReportTour_2024-08-19.pdf</ext-link></comment></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Natale</surname><given-names>P</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chow</surname><given-names>CK</given-names> </name><etal/></person-group><article-title>Patient experiences of continuous glucose monitoring and sensor-augmented insulin pump therapy for diabetes: a systematic review of qualitative studies</article-title><source>J Diabetes</source><year>2023</year><month>12</month><volume>15</volume><issue>12</issue><fpage>1048</fpage><lpage>1069</lpage><pub-id pub-id-type="doi">10.1111/1753-0407.13454</pub-id><pub-id pub-id-type="medline">37551735</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lawton</surname><given-names>J</given-names> </name><name name-style="western"><surname>Blackburn</surname><given-names>M</given-names> </name><name name-style="western"><surname>Allen</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Patients&#x2019; and caregivers&#x2019; experiences of using continuous glucose monitoring to support diabetes self-management: qualitative study</article-title><source>BMC Endocr Disord</source><year>2018</year><month>02</month><day>20</day><volume>18</volume><issue>1</issue><fpage>12</fpage><pub-id pub-id-type="doi">10.1186/s12902-018-0239-1</pub-id><pub-id pub-id-type="medline">29458348</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kongdee</surname><given-names>R</given-names> </name><name name-style="western"><surname>Parsia</surname><given-names>B</given-names> </name><name name-style="western"><surname>Thabit</surname><given-names>H</given-names> </name><name name-style="western"><surname>Harper</surname><given-names>S</given-names> </name></person-group><article-title>Glucose interpretation meaning and action (GIMA): insights to blood glucose user interface interpretation in type 1 diabetes</article-title><source>Digit Health</source><year>2025</year><volume>11</volume><fpage>20552076251332580</fpage><pub-id pub-id-type="doi">10.1177/20552076251332580</pub-id><pub-id pub-id-type="medline">40351844</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="web"><article-title>Chapter 3: diabetes distress</article-title><source>Diabetes UK</source><access-date>2026-03-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.diabetes.org.uk/for-professionals/improving-care/good-practice/psychological-care/emotional-health-professionals-guide/chapter-3-diabetes-distress">https://www.diabetes.org.uk/for-professionals/improving-care/good-practice/psychological-care/emotional-health-professionals-guide/chapter-3-diabetes-distress</ext-link></comment></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Holt</surname><given-names>RIG</given-names> </name><name name-style="western"><surname>de Groot</surname><given-names>M</given-names> </name><name name-style="western"><surname>Golden</surname><given-names>SH</given-names> </name></person-group><article-title>Diabetes and depression</article-title><source>Curr Diab Rep</source><year>2014</year><month>06</month><volume>14</volume><issue>6</issue><fpage>491</fpage><pub-id pub-id-type="doi">10.1007/s11892-014-0491-3</pub-id><pub-id pub-id-type="medline">24743941</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="web"><article-title>Links between diabetes and depression</article-title><source>Diabetes UK</source><access-date>2026-03-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.diabetes.org.uk/living-with-diabetes/emotional-wellbeing/depression">https://www.diabetes.org.uk/living-with-diabetes/emotional-wellbeing/depression</ext-link></comment></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sheng</surname><given-names>B</given-names> </name><name name-style="western"><surname>Guan</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Lim</surname><given-names>LL</given-names> </name><etal/></person-group><article-title>Large language models for diabetes care: potentials and prospects</article-title><source>Sci Bull (Beijing)</source><year>2024</year><month>03</month><day>15</day><volume>69</volume><issue>5</issue><fpage>583</fpage><lpage>588</lpage><pub-id pub-id-type="doi">10.1016/j.scib.2024.01.004</pub-id><pub-id pub-id-type="medline">38220476</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Large language models in clinical trials: applications, technical advances, and future directions</article-title><source>BMC Med</source><year>2025</year><month>10</month><day>14</day><volume>23</volume><issue>1</issue><fpage>563</fpage><pub-id pub-id-type="doi">10.1186/s12916-025-04348-9</pub-id><pub-id pub-id-type="medline">41088200</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Lai</surname><given-names>A</given-names> </name><name name-style="western"><surname>Thygesen</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Farrington</surname><given-names>J</given-names> </name><name name-style="western"><surname>Keen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Li</surname><given-names>K</given-names> </name></person-group><article-title>Large language models for mental health applications: systematic review</article-title><source>JMIR Ment Health</source><year>2024</year><month>10</month><day>18</day><volume>11</volume><fpage>e57400</fpage><pub-id pub-id-type="doi">10.2196/57400</pub-id><pub-id pub-id-type="medline">39423368</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zeng</surname><given-names>J</given-names> </name><name name-style="western"><surname>Qi</surname><given-names>W</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Embracing the future of medical education with large language model-based virtual patients: scoping review</article-title><source>J Med Internet Res</source><year>2025</year><month>11</month><day>13</day><volume>27</volume><fpage>e79091</fpage><pub-id pub-id-type="doi">10.2196/79091</pub-id><pub-id pub-id-type="medline">41232097</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gong</surname><given-names>EJ</given-names> </name><name name-style="western"><surname>Bang</surname><given-names>CS</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Baik</surname><given-names>GH</given-names> </name></person-group><article-title>Knowledge-practice performance gap in clinical large language models: systematic review of 39 benchmarks</article-title><source>J Med Internet Res</source><year>2025</year><month>12</month><day>1</day><volume>27</volume><fpage>e84120</fpage><pub-id pub-id-type="doi">10.2196/84120</pub-id><pub-id pub-id-type="medline">41325597</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Lai</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ive</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Development and evaluation of hopebot: an LLM-based chatbot for structured and interactive PHQ-9 depression screening</article-title><source>arXiv</source><access-date>2026-03-17</access-date><comment>Preprint posted online on  Jan 14, 2026</comment><comment><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.48550/arXiv.2507.05984">https://doi.org/10.48550/arXiv.2507.05984</ext-link></comment><pub-id pub-id-type="doi">10.21203/rs.3.rs-6976450/v1</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kelly</surname><given-names>A</given-names> </name><name name-style="western"><surname>Noctor</surname><given-names>E</given-names> </name><name name-style="western"><surname>Ryan</surname><given-names>L</given-names> </name><name name-style="western"><surname>van de Ven</surname><given-names>P</given-names> </name></person-group><article-title>The effectiveness of a custom AI chatbot for type 2 diabetes mellitus health literacy: development and evaluation study</article-title><source>J Med Internet Res</source><year>2025</year><month>05</month><day>5</day><volume>27</volume><fpage>e70131</fpage><pub-id pub-id-type="doi">10.2196/70131</pub-id><pub-id pub-id-type="medline">40324160</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Parameswaran</surname><given-names>V</given-names> </name><name name-style="western"><surname>Bernard</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bernard</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Evaluating large language models and retrieval-augmented generation enhancement for delivering guideline-adherent nutrition information for cardiovascular disease prevention: cross-sectional study</article-title><source>J Med Internet Res</source><year>2025</year><month>10</month><day>7</day><volume>27</volume><fpage>e78625</fpage><pub-id pub-id-type="doi">10.2196/78625</pub-id><pub-id pub-id-type="medline">41057043</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jeon</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>EH</given-names> </name><etal/></person-group><article-title>Generative AI chatbot for diabetes management: formative 2-part qualitative study using DTalksBot involving patients and clinicians</article-title><source>JMIR Form Res</source><year>2025</year><month>11</month><day>12</day><volume>9</volume><fpage>e72553</fpage><pub-id pub-id-type="doi">10.2196/72553</pub-id><pub-id pub-id-type="medline">41223424</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ge</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Application of chatbots to help patients self-manage diabetes: systematic review and meta-analysis</article-title><source>J Med Internet Res</source><year>2024</year><month>12</month><day>3</day><volume>26</volume><fpage>e60380</fpage><pub-id pub-id-type="doi">10.2196/60380</pub-id><pub-id pub-id-type="medline">39626235</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Healey</surname><given-names>E</given-names> </name><name name-style="western"><surname>Kohane</surname><given-names>IS</given-names> </name></person-group><article-title>LLM-CGM: a benchmark for large language model-enabled querying of continuous glucose monitoring data for conversational diabetes management</article-title><source>Pac Symp Biocomput</source><year>2025</year><volume>30</volume><fpage>82</fpage><lpage>93</lpage><pub-id pub-id-type="doi">10.1142/9789819807024_0007</pub-id><pub-id pub-id-type="medline">39670363</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Healey</surname><given-names>E</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>ALM</given-names> </name><name name-style="western"><surname>Flint</surname><given-names>KL</given-names> </name><name name-style="western"><surname>Ruiz</surname><given-names>JL</given-names> </name><name name-style="western"><surname>Kohane</surname><given-names>IS</given-names> </name></person-group><article-title>A case study on using a large language model to analyze continuous glucose monitoring data</article-title><source>Sci Rep</source><year>2025</year><month>01</month><day>7</day><volume>15</volume><issue>1</issue><fpage>1143</fpage><pub-id pub-id-type="doi">10.1038/s41598-024-84003-0</pub-id><pub-id pub-id-type="medline">39774031</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Greenfield</surname><given-names>M</given-names> </name><name name-style="western"><surname>Stuber</surname><given-names>D</given-names> </name><name name-style="western"><surname>Stegman-Barber</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Diabetes education and support tele-visit needs differ in duration, content, and satisfaction in older versus younger adults</article-title><source>Telemed Rep</source><year>2022</year><volume>3</volume><issue>1</issue><fpage>107</fpage><lpage>116</lpage><pub-id pub-id-type="doi">10.1089/tmr.2022.0007</pub-id><pub-id pub-id-type="medline">35720451</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gholamzadeh</surname><given-names>M</given-names> </name><name name-style="western"><surname>Abtahi</surname><given-names>H</given-names> </name><name name-style="western"><surname>Ghazisaeeidi</surname><given-names>M</given-names> </name></person-group><article-title>Applied techniques for putting pre-visit planning in clinical practice to empower patient-centered care in the pandemic era: a systematic review and framework suggestion</article-title><source>BMC Health Serv Res</source><year>2021</year><month>05</month><day>13</day><volume>21</volume><issue>1</issue><fpage>458</fpage><pub-id pub-id-type="doi">10.1186/s12913-021-06456-7</pub-id><pub-id pub-id-type="medline">33985502</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bellido</surname><given-names>V</given-names> </name><name name-style="western"><surname>Aguilera</surname><given-names>E</given-names> </name><name name-style="western"><surname>Cardona-Hernandez</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Expert recommendations for using time-in-range and other continuous glucose monitoring metrics to achieve patient-centered glycemic control in people with diabetes</article-title><source>J Diabetes Sci Technol</source><year>2023</year><month>09</month><volume>17</volume><issue>5</issue><fpage>1326</fpage><lpage>1336</lpage><pub-id pub-id-type="doi">10.1177/19322968221088601</pub-id><pub-id pub-id-type="medline">35470692</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="web"><article-title>What approvals and decisions do i need?</article-title><source>Health Research Authority</source><access-date>2026-04-02</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.hra.nhs.uk/approvals-amendments/what-approvals-do-i-need">https://www.hra.nhs.uk/approvals-amendments/what-approvals-do-i-need</ext-link></comment></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Zhu</surname><given-names>J</given-names> </name></person-group><article-title>Diabetes datasets-ShanghaiT1DM and ShanghaiT2DM [dataset]</article-title><year>2022</year><publisher-name>Figshare</publisher-name><pub-id pub-id-type="doi">10.6084/m9.figshare.20444397.v3</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="web"><article-title>OhioT1DM dataset</article-title><source>University of North Carolina at Charlotte</source><access-date>2026-03-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://webpages.charlotte.edu/rbunescu/data/ohiot1dm/OhioT1DM-dataset.html">https://webpages.charlotte.edu/rbunescu/data/ohiot1dm/OhioT1DM-dataset.html</ext-link></comment></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Danne</surname><given-names>T</given-names> </name><name name-style="western"><surname>Nimri</surname><given-names>R</given-names> </name><name name-style="western"><surname>Battelino</surname><given-names>T</given-names> </name><etal/></person-group><article-title>International consensus on use of continuous glucose monitoring</article-title><source>Diabetes Care</source><year>2017</year><month>12</month><volume>40</volume><issue>12</issue><fpage>1631</fpage><lpage>1640</lpage><pub-id pub-id-type="doi">10.2337/dc17-1600</pub-id><pub-id pub-id-type="medline">29162583</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Doupis</surname><given-names>J</given-names> </name><name name-style="western"><surname>Horton</surname><given-names>ES</given-names> </name></person-group><article-title>Utilizing the new glucometrics: a practical guide to ambulatory glucose profile interpretation</article-title><source>touchREV Endocrinol</source><year>2022</year><month>06</month><volume>18</volume><issue>1</issue><fpage>20</fpage><lpage>26</lpage><pub-id pub-id-type="doi">10.17925/EE.2022.18.1.20</pub-id><pub-id pub-id-type="medline">35949362</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Szmuilowicz</surname><given-names>ED</given-names> </name><name name-style="western"><surname>Aleppo</surname><given-names>G</given-names> </name></person-group><article-title>Stepwise approach to continuous glucose monitoring interpretation for internists and family physicians</article-title><source>Postgrad Med</source><year>2022</year><month>11</month><volume>134</volume><issue>8</issue><fpage>743</fpage><lpage>751</lpage><pub-id pub-id-type="doi">10.1080/00325481.2022.2110507</pub-id><pub-id pub-id-type="medline">35930313</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="web"><article-title>Quick guide: interpreting CGM data</article-title><source>Diabetes on the net</source><year>2020</year><access-date>2026-03-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://diabetesonthenet.com/journal-diabetes-nursing/quick-guide-interpreting-cgm-data/">https://diabetesonthenet.com/journal-diabetes-nursing/quick-guide-interpreting-cgm-data/</ext-link></comment></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="web"><article-title>Quality statement 4: continuous glucose monitoring for adults who use insulin and need help monitoring their blood glucose</article-title><source>NICE</source><year>2023</year><access-date>2026-03-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.nice.org.uk/guidance/qs209/chapter/Quality-statement-4-Continuous-glucose-monitoring-for-adults-who-use-insulin-and-need-help-monitoring-their-blood-glucose">https://www.nice.org.uk/guidance/qs209/chapter/Quality-statement-4-Continuous-glucose-monitoring-for-adults-who-use-insulin-and-need-help-monitoring-their-blood-glucose</ext-link></comment></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="web"><article-title>AGP report</article-title><source>Accu-Chek</source><access-date>2026-03-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.accu-chek.co.uk/training/cgm/agp-report">https://www.accu-chek.co.uk/training/cgm/agp-report</ext-link></comment></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="web"><article-title>Managing diabetes</article-title><source>National Institute of Diabetes and Digestive and Kidney Diseases</source><access-date>2026-03-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.niddk.nih.gov/health-information/diabetes/overview/managing-diabetes">https://www.niddk.nih.gov/health-information/diabetes/overview/managing-diabetes</ext-link></comment></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="web"><article-title>Diabetes: what it is, causes, symptoms, treatment and types</article-title><source>Cleveland Clinic</source><year>2023</year><access-date>2023-02-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://my.clevelandclinic.org/health/diseases/7104-diabetes">https://my.clevelandclinic.org/health/diseases/7104-diabetes</ext-link></comment></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Battelino</surname><given-names>T</given-names> </name><name name-style="western"><surname>Danne</surname><given-names>T</given-names> </name><name name-style="western"><surname>Bergenstal</surname><given-names>RM</given-names> </name><etal/></person-group><article-title>Clinical targets for continuous glucose monitoring data interpretation: recommendations from the international consensus on time in range</article-title><source>Diabetes Care</source><year>2019</year><month>08</month><volume>42</volume><issue>8</issue><fpage>1593</fpage><lpage>1603</lpage><pub-id pub-id-type="doi">10.2337/dci19-0028</pub-id><pub-id pub-id-type="medline">31177185</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="web"><article-title>Type 2 diabetes in adults: management</article-title><source>NICE</source><year>2015</year><month>12</month><day>2</day><access-date>2026-03-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.nice.org.uk/guidance/ng28">https://www.nice.org.uk/guidance/ng28</ext-link></comment></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>ElSayed</surname><given-names>NA</given-names> </name><name name-style="western"><surname>Aleppo</surname><given-names>G</given-names> </name><name name-style="western"><surname>Aroda</surname><given-names>VR</given-names> </name><etal/></person-group><article-title>5. Facilitating positive health behaviors and well-being to improve health outcomes: standards of care in diabetes-2023</article-title><source>Diabetes Care</source><year>2023</year><month>01</month><day>1</day><volume>46</volume><issue>Supple 1</issue><fpage>S68</fpage><lpage>S96</lpage><pub-id pub-id-type="doi">10.2337/dc23-S005</pub-id><pub-id pub-id-type="medline">36507648</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="web"><article-title>What is diabetes distress and burnout?</article-title><source>Diabetes UK</source><access-date>2026-03-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.diabetes.org.uk/living-with-diabetes/emotional-wellbeing/diabetes-burnout">https://www.diabetes.org.uk/living-with-diabetes/emotional-wellbeing/diabetes-burnout</ext-link></comment></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="web"><article-title>Diabetes and your emotions</article-title><source>Diabetes UK</source><access-date>2026-03-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.diabetes.org.uk/living-with-diabetes/emotional-wellbeing">https://www.diabetes.org.uk/living-with-diabetes/emotional-wellbeing</ext-link></comment></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="web"><article-title>10 tips to ease diabetes stress</article-title><source>American Diabetes Association</source><access-date>2026-03-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://diabetes.org/health-wellness/mental-health/ease-diabetes-care-stress">https://diabetes.org/health-wellness/mental-health/ease-diabetes-care-stress</ext-link></comment></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zeng</surname><given-names>G</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Ju</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>MedDialog: large-scale medical dialogue datasets</article-title><year>2020</year><access-date>2026-07-14</access-date><conf-name>Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP)</conf-name><conf-date>Nov 16-20, 2020</conf-date><conf-loc>Online</conf-loc><fpage>9241</fpage><lpage>9250</lpage><pub-id pub-id-type="doi">10.18653/v1/2020.emnlp-main.743</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="web"><article-title>OpenAIEmbeddings integration</article-title><source>LangChain</source><access-date>2026-03-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://docs.langchain.com/oss/python/integrations/embeddings/openai">https://docs.langchain.com/oss/python/integrations/embeddings/openai</ext-link></comment></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="web"><article-title>Faiss</article-title><source>Meta AI</source><access-date>2026-03-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://ai.meta.com/tools/faiss/">https://ai.meta.com/tools/faiss/</ext-link></comment></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Douze</surname><given-names>M</given-names> </name><name name-style="western"><surname>Guzhva</surname><given-names>A</given-names> </name><name name-style="western"><surname>Deng</surname><given-names>C</given-names> </name><etal/></person-group><article-title>The Faiss library</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 23, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2401.08281</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Koo</surname><given-names>TK</given-names> </name><name name-style="western"><surname>Li</surname><given-names>MY</given-names> </name></person-group><article-title>A guideline of selecting and reporting intraclass correlation coefficients for reliability research</article-title><source>J Chiropr Med</source><year>2016</year><month>06</month><volume>15</volume><issue>2</issue><fpage>155</fpage><lpage>163</lpage><pub-id pub-id-type="doi">10.1016/j.jcm.2016.02.012</pub-id><pub-id pub-id-type="medline">27330520</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>FL</given-names> </name></person-group><article-title>Using cluster bootstrapping to analyze nested data with a few clusters</article-title><source>Educ Psychol Meas</source><year>2018</year><month>04</month><volume>78</volume><issue>2</issue><fpage>297</fpage><lpage>318</lpage><pub-id pub-id-type="doi">10.1177/0013164416678980</pub-id><pub-id pub-id-type="medline">29795957</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Silveira</surname><given-names>L da</given-names> </name><name name-style="western"><surname>Ferreira</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Patino</surname><given-names>CM</given-names> </name></person-group><article-title>Mixed-effects model: a useful statistical tool for longitudinal and cluster studies</article-title><source>J Bras Pneumol</source><year>2023</year><month>05</month><day>15</day><volume>49</volume><issue>2</issue><fpage>e20230137</fpage><pub-id pub-id-type="doi">10.36416/1806-3756/e20230137</pub-id><pub-id pub-id-type="medline">37194822</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ludbrook</surname><given-names>J</given-names> </name></person-group><article-title>Should we use one-sided or two-sided P values in tests of significance?</article-title><source>Clin Exp Pharmacol Physiol</source><year>2013</year><month>06</month><volume>40</volume><issue>6</issue><fpage>357</fpage><lpage>361</lpage><pub-id pub-id-type="doi">10.1111/1440-1681.12086</pub-id><pub-id pub-id-type="medline">23551169</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mishra</surname><given-names>P</given-names> </name><name name-style="western"><surname>Pandey</surname><given-names>CM</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>U</given-names> </name><name name-style="western"><surname>Gupta</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sahu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Keshri</surname><given-names>A</given-names> </name></person-group><article-title>Descriptive statistics and normality tests for statistical data</article-title><source>Ann Card Anaesth</source><year>2019</year><volume>22</volume><issue>1</issue><fpage>67</fpage><lpage>72</lpage><pub-id pub-id-type="doi">10.4103/aca.ACA_157_18</pub-id><pub-id pub-id-type="medline">30648682</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="web"><article-title>Wilcoxon signed ranks test</article-title><source>ScienceDirect Topics</source><access-date>2026-03-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.sciencedirect.com/topics/medicine-and-dentistry/wilcoxon-signed-ranks-test/">https://www.sciencedirect.com/topics/medicine-and-dentistry/wilcoxon-signed-ranks-test/</ext-link></comment></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Drton</surname><given-names>M</given-names> </name><name name-style="western"><surname>Xiao</surname><given-names>H</given-names> </name></person-group><article-title>Wald tests of singular hypotheses</article-title><source>Bernoulli</source><year>2016</year><volume>22</volume><issue>1</issue><fpage>38</fpage><lpage>59</lpage><pub-id pub-id-type="doi">10.3150/14-BEJ620</pub-id></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>French</surname><given-names>RM</given-names> </name></person-group><article-title>The turing test: the first 50 years</article-title><source>Trends Cogn Sci</source><year>2000</year><month>03</month><volume>4</volume><issue>3</issue><fpage>115</fpage><lpage>122</lpage><pub-id pub-id-type="doi">10.1016/s1364-6613(00)01453-4</pub-id><pub-id pub-id-type="medline">10689346</pub-id></nlm-citation></ref><ref id="ref64"><label>64</label><nlm-citation citation-type="web"><article-title>The binomial test</article-title><source>Technology Networks</source><year>2024</year><access-date>2026-03-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="http://www.technologynetworks.com/informatics/articles/the-binomial-test-366022">http://www.technologynetworks.com/informatics/articles/the-binomial-test-366022</ext-link></comment></nlm-citation></ref><ref id="ref65"><label>65</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vasey</surname><given-names>B</given-names> </name><name name-style="western"><surname>Nagendran</surname><given-names>M</given-names> </name><name name-style="western"><surname>Campbell</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Reporting guideline for the early-stage clinical evaluation of decision support systems driven by artificial intelligence: DECIDE-AI</article-title><source>Nat Med</source><year>2022</year><month>05</month><volume>28</volume><issue>5</issue><fpage>924</fpage><lpage>933</lpage><pub-id pub-id-type="doi">10.1038/s41591-022-01772-9</pub-id><pub-id pub-id-type="medline">35585198</pub-id></nlm-citation></ref><ref id="ref66"><label>66</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>MacFarland</surname><given-names>TW</given-names> </name><name name-style="western"><surname>Yates</surname><given-names>JM</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>MacFarland</surname><given-names>TW</given-names> </name><name name-style="western"><surname>Yates</surname><given-names>JM</given-names> </name></person-group><article-title>Kruskal-wallis h-test for oneway analysis of variance (ANOVA) by ranks</article-title><source>Introduction to Nonparametric Statistics for the Biological Sciences Using R</source><year>2016</year><publisher-name>Springer International Publishing</publisher-name><fpage>177</fpage><lpage>211</lpage><pub-id pub-id-type="doi">10.1007/978-3-319-30634-6_6</pub-id><pub-id pub-id-type="other">978-3-319-30633-9</pub-id></nlm-citation></ref><ref id="ref67"><label>67</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Q</given-names> </name><etal/></person-group><article-title>Foundation models and intelligent decision-making: progress, challenges, and perspectives</article-title><source>Innovation (Camb)</source><year>2025</year><month>06</month><day>2</day><volume>6</volume><issue>6</issue><fpage>100948</fpage><pub-id pub-id-type="doi">10.1016/j.xinn.2025.100948</pub-id><pub-id pub-id-type="medline">40528892</pub-id></nlm-citation></ref><ref id="ref68"><label>68</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Rezaei</surname><given-names>SJ</given-names> </name><etal/></person-group><article-title>Artificial intelligence tools in supporting healthcare professionals for tailored patient care</article-title><source>NPJ Digit Med</source><year>2025</year><month>04</month><day>16</day><volume>8</volume><issue>1</issue><fpage>210</fpage><pub-id pub-id-type="doi">10.1038/s41746-025-01604-3</pub-id><pub-id pub-id-type="medline">40240489</pub-id></nlm-citation></ref><ref id="ref69"><label>69</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gaoudam</surname><given-names>N</given-names> </name><name name-style="western"><surname>Sakhamudi</surname><given-names>SK</given-names> </name><name name-style="western"><surname>Kamal</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Wearable devices and AI-driven remote monitoring in cardiovascular medicine: a narrative review</article-title><source>Cureus</source><year>2025</year><month>08</month><volume>17</volume><issue>8</issue><fpage>e90208</fpage><pub-id pub-id-type="doi">10.7759/cureus.90208</pub-id><pub-id pub-id-type="medline">40964568</pub-id></nlm-citation></ref><ref id="ref70"><label>70</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Canali</surname><given-names>S</given-names> </name><name name-style="western"><surname>Schiaffonati</surname><given-names>V</given-names> </name><name name-style="western"><surname>Aliverti</surname><given-names>A</given-names> </name></person-group><article-title>Challenges and recommendations for wearable devices in digital health: data quality, interoperability, health equity, fairness</article-title><source>PLoS Digit Health</source><year>2022</year><month>10</month><volume>1</volume><issue>10</issue><fpage>e0000104</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000104</pub-id><pub-id pub-id-type="medline">36812619</pub-id></nlm-citation></ref><ref id="ref71"><label>71</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bent</surname><given-names>B</given-names> </name><name name-style="western"><surname>Goldstein</surname><given-names>BA</given-names> </name><name name-style="western"><surname>Kibbe</surname><given-names>WA</given-names> </name><name name-style="western"><surname>Dunn</surname><given-names>JP</given-names> </name></person-group><article-title>Investigating sources of inaccuracy in wearable optical heart rate sensors</article-title><source>NPJ Digit Med</source><year>2020</year><volume>3</volume><fpage>18</fpage><pub-id pub-id-type="doi">10.1038/s41746-020-0226-6</pub-id><pub-id pub-id-type="medline">32047863</pub-id></nlm-citation></ref><ref id="ref72"><label>72</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>WJ</given-names> </name><name name-style="western"><surname>Li</surname><given-names>LZ</given-names> </name></person-group><article-title>Artificial intelligence in mobile health applications: a comprehensive review of its role in diabetes care</article-title><source>World J Methodol</source><year>2026</year><month>03</month><day>20</day><volume>16</volume><issue>1</issue><fpage>107488</fpage><pub-id pub-id-type="doi">10.5662/wjm.v16.i1.107488</pub-id><pub-id pub-id-type="medline">41809156</pub-id></nlm-citation></ref><ref id="ref73"><label>73</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schrangl</surname><given-names>P</given-names> </name><name name-style="western"><surname>Reiterer</surname><given-names>F</given-names> </name><name name-style="western"><surname>Heinemann</surname><given-names>L</given-names> </name><name name-style="western"><surname>Freckmann</surname><given-names>G</given-names> </name><name name-style="western"><surname>Del Re</surname><given-names>L</given-names> </name></person-group><article-title>Limits to the evaluation of the accuracy of continuous glucose monitoring systems by clinical trials</article-title><source>Biosensors (Basel)</source><year>2018</year><month>05</month><day>18</day><volume>8</volume><issue>2</issue><fpage>50</fpage><pub-id pub-id-type="doi">10.3390/bios8020050</pub-id><pub-id pub-id-type="medline">29783669</pub-id></nlm-citation></ref><ref id="ref74"><label>74</label><nlm-citation citation-type="web"><article-title>GLP-1 agonists</article-title><source>Diabetes UK</source><access-date>2026-03-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.diabetes.org.uk/about-diabetes/looking-after-diabetes/treatments/tablets-and-medication/glp-1">https://www.diabetes.org.uk/about-diabetes/looking-after-diabetes/treatments/tablets-and-medication/glp-1</ext-link></comment></nlm-citation></ref><ref id="ref75"><label>75</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Walter</surname><given-names>SR</given-names> </name><name name-style="western"><surname>Dunsmuir</surname><given-names>WTM</given-names> </name><name name-style="western"><surname>Westbrook</surname><given-names>JI</given-names> </name></person-group><article-title>Inter-observer agreement and reliability assessment for observational studies of clinical work</article-title><source>J Biomed Inform</source><year>2019</year><month>12</month><volume>100</volume><fpage>103317</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2019.103317</pub-id><pub-id pub-id-type="medline">31654801</pub-id></nlm-citation></ref><ref id="ref76"><label>76</label><nlm-citation citation-type="web"><article-title>How many people in the UK have diabetes?</article-title><source>Diabetes UK</source><access-date>2026-05-30</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.diabetes.org.uk/about-us/about-the-charity/our-strategy/statistics">https://www.diabetes.org.uk/about-us/about-the-charity/our-strategy/statistics</ext-link></comment></nlm-citation></ref><ref id="ref77"><label>77</label><nlm-citation citation-type="web"><article-title>Quality statement 3: continuous glucose monitoring for adults on multiple daily insulin injections who cannot self-monitor using capillary blood glucose monitoring in: type 2 diabetes in adults NICE quality standard QS209</article-title><source>National Institute for Health and Care Excellence</source><year>2023</year><access-date>2026-06-06</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.nice.org.uk/guidance/qs209/chapter/Quality-statement-3-Continuous-glucose-monitoring-for-adults-on-multiple-daily-insulin-injections-who-cannot-self-monitor-using-capillary-blood-glucose-monitoring">https://www.nice.org.uk/guidance/qs209/chapter/Quality-statement-3-Continuous-glucose-monitoring-for-adults-on-multiple-daily-insulin-injections-who-cannot-self-monitor-using-capillary-blood-glucose-monitoring</ext-link></comment></nlm-citation></ref><ref id="ref78"><label>78</label><nlm-citation citation-type="web"><article-title>National diabetes audit core report 1: care processes and treatment targets 2024-25 underlying data</article-title><source>NHS England Digital</source><year>2026</year><access-date>2026-06-06</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://digital.nhs.uk/data-and-information/publications/statistical/national-diabetes-audit/report-1-cp-and-tt-data-release-2024-25/nda-report-1-cp-and-tt-data-release-2024-25">https://digital.nhs.uk/data-and-information/publications/statistical/national-diabetes-audit/report-1-cp-and-tt-data-release-2024-25/nda-report-1-cp-and-tt-data-release-2024-25</ext-link></comment></nlm-citation></ref><ref id="ref79"><label>79</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Lai</surname><given-names>A</given-names> </name><name name-style="western"><surname>Deng</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Li</surname><given-names>K</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Xie</surname><given-names>X</given-names> </name><name name-style="western"><surname>Styles</surname><given-names>I</given-names> </name><name name-style="western"><surname>Powathil</surname><given-names>G</given-names> </name><name name-style="western"><surname>Ceccarelli</surname><given-names>M</given-names> </name></person-group><article-title>Evaluating the feasibility and acceptability of a GPT-based chatbot for depression screening: a mixed-methods study</article-title><source>Artificial Intelligence in Healthcare AIiH 2024 Lecture Notes in Computer Science</source><year>2024</year><publisher-name>Springer</publisher-name><fpage>258</fpage><lpage>271</lpage><pub-id pub-id-type="doi">10.1007/978-3-031-67278-1_20</pub-id></nlm-citation></ref><ref id="ref80"><label>80</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>Z</given-names> </name></person-group><article-title>Diabetes-chatbot</article-title><source>GitHub</source><year>2026</year><access-date>2026-07-15</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/candiceguo0528/Diabetes-chatbot">https://github.com/candiceguo0528/Diabetes-chatbot</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Example patient profile.</p><media xlink:href="jmir_v28i1e98519_app1.docx" xlink:title="DOCX File, 145 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Prompt to build the conversational agent.</p><media xlink:href="jmir_v28i1e98519_app2.docx" xlink:title="DOCX File, 19 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Full question bank.</p><media xlink:href="jmir_v28i1e98519_app3.docx" xlink:title="DOCX File, 20 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Clinician case assignments, question-ID coverage, and reviewed cases &#x0026; Frequency of Each Question ID Across All 12 Cases (N=144).</p><media xlink:href="jmir_v28i1e98519_app4.docx" xlink:title="DOCX File, 19 KB"/></supplementary-material><supplementary-material id="app5"><label>Multimedia Appendix 5</label><p>Retrieval audit of top-ranked retrieved segments and their relationship to archived clinical assistant responses.</p><media xlink:href="jmir_v28i1e98519_app5.docx" xlink:title="DOCX File, 31 KB"/></supplementary-material><supplementary-material id="app6"><label>Multimedia Appendix 6 </label><p>Association between response length and quality ratings.</p><media xlink:href="jmir_v28i1e98519_app6.docx" xlink:title="DOCX File, 18 KB"/></supplementary-material><supplementary-material id="app7"><label>Multimedia Appendix 7 </label><p>Domain-specific mixed effects model results comparing CA and clinician responses.</p><media xlink:href="jmir_v28i1e98519_app7.docx" xlink:title="DOCX File, 17 KB"/></supplementary-material><supplementary-material id="app8"><label>Multimedia Appendix 8 </label><p>Domain- and dimension-specific quality ratings (mean &#x00B1; SD).</p><media xlink:href="jmir_v28i1e98519_app8.docx" xlink:title="DOCX File, 19 KB"/></supplementary-material><supplementary-material id="app9"><label>Multimedia Appendix 9 </label><p>Rater-level distribution of overall quality scores stratified by perceived source.</p><media xlink:href="jmir_v28i1e98519_app9.docx" xlink:title="DOCX File, 109 KB"/></supplementary-material><supplementary-material id="app10"><label>Checklist 1</label><p>DECIDE-AI checklist.</p><media xlink:href="jmir_v28i1e98519_app10.docx" xlink:title="DOCX File, 26 KB"/></supplementary-material></app-group></back></article>