<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e94140</article-id><article-id pub-id-type="doi">10.2196/94140</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Effects of a Safety User Interface Bundle on Verification Intentions in Generative AI Chat Use Among Older Chinese Adults: Randomized Vignette Survey</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Yu</surname><given-names>Jun'an</given-names></name><degrees>MInfoTech</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chen</surname><given-names>Jun</given-names></name><degrees>MSMHSM</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ren</surname><given-names>Anjie</given-names></name><degrees>BS</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Duan</surname><given-names>Hui</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Meng</surname><given-names>Hua</given-names></name><degrees>MEng</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Gao</surname><given-names>Zhuo</given-names></name><degrees>LLM</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib></contrib-group><aff id="aff1"><institution>Faculty of Science, University of Auckland</institution><addr-line>Auckland</addr-line><country>New Zealand</country></aff><aff id="aff2"><institution>Medical Services Management Department, Peking University People&#x2019;s Hospital (PKUPH)</institution><addr-line>Beijing</addr-line><country>China</country></aff><aff id="aff3"><institution>Healthcare &#x0026; Education Research Center, Chengdu Gongyun Education &#x0026; Management Research Institute</institution><addr-line>Chengdu</addr-line><country>China</country></aff><aff id="aff4"><institution>School of Public Administration and Policy, Renmin University of China</institution><addr-line>Beijing</addr-line><country>China</country></aff><aff id="aff5"><institution>Department of Economics and Management, Sichuan University of Architectural Technology</institution><addr-line>Chengdu</addr-line><country>China</country></aff><aff id="aff6"><institution>Department of Human Resource Management, Beijing Geriatric Hospital</institution><addr-line>118 Wenquan Road, Haidian District</addr-line><addr-line>Beijing</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Katz</surname><given-names>Adi</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Marshall</surname><given-names>Robert</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Zhuo Gao, LLM, Department of Human Resource Management, Beijing Geriatric Hospital, 118 Wenquan Road, Haidian District, Beijing, 100095, China, 86 15201400966; <email>bjghgaozhuo@163.com</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>14</day><month>8</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e94140</elocation-id><history><date date-type="received"><day>25</day><month>02</month><year>2026</year></date><date date-type="rev-recd"><day>18</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>21</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Jun'an Yu, Jun Chen, Anjie Ren, Hui Duan, Hua Meng, Zhuo Gao. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 14.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e94140"/><abstract><sec><title>Background</title><p>Generative AI chat systems are increasingly used for everyday information seeking, but plausible errors and omissions can mislead users when outputs are accepted without scrutiny. Interface-level safety cues may help users calibrate trust and engage in verification; yet, evidence in older Chinese adults remains limited.</p></sec><sec><title>Objective</title><p>This study aimed to test whether adding a safety user interface (UI) bundle to a generative AI chat interface increases verification intention among older Chinese adults and to examine selected secondary outcomes, including reliance intention, trust calibration, perceived trustworthiness, comprehension, usability/readability, cognitive load, and a behavioral proxy of verification.</p></sec><sec sec-type="methods"><title>Methods</title><p>We conducted a cross-sectional survey with an embedded randomized UI vignette experiment between May 22, 2025, and September 3, 2025. Chinese adults aged &#x2265;60 years were recruited through community sites, outpatient clinic waiting areas, and WeChat (Tencent Holdings Ltd) groups, and randomized 1:1 to view screenshots of a baseline chat UI or a safety UI bundle containing generic source-label cues, and an uncertainty and verification nudge. Each participant completed 2 scenarios (service/travel decision and general well-being related to sleep/fatigue), followed by measures of verification intention (primary), reliance intention, trust calibration index, comprehension (0&#x2010;8), perceived trustworthiness, usability/readability, cognitive load (0&#x2010;10), manipulation checks, and a behavioral proxy (expanding optional &#x201C;source information&#x201D;). Analyses used intention-to-treat regression models with covariate adjustment.</p></sec><sec sec-type="results"><title>Results</title><p>Of 214 consenting respondents who started the survey, 200 were included in the analysis (100 per arm). The safety UI bundle increased verification intention (mean 4.72, SD 0.63 vs 4.41, SD 0.59 on a 7-point scale; adjusted &#x03B2;=0.293, 95% CI 0.128-0.457; <italic>P</italic>&#x003C;.001). Reliance intention did not increase (mean 4.97, SD 0.54 vs 5.03, SD 0.58; adjusted &#x03B2;=&#x2212;0.105, 95% CI &#x2212;0.239 to 0.029; <italic>P</italic>=.13). Trust calibration improved (trust calibration index: mean &#x2212;0.29, SD 1.43 vs 0.29, SD 1.43; adjusted &#x03B2;=&#x2212;0.567, 95% CI &#x2212;1.005 to &#x2212;0.129; P=.01). Expansion of optional source information was numerically higher, although the adjusted CI included the null (42% vs 27%; adjusted odds ratio [OR]=1.76, 95% CI 0.95-3.27; <italic>P</italic>=.07). Comprehension remained high and similar across arms (mean 6.33, SD 1.14 vs 6.32, SD 1.08; adjusted &#x03B2;=&#x2212;0.132, 95% CI &#x2212;0.428 to 0.163; <italic>P</italic>=.38). Perceived trustworthiness was modestly lower in the Safety UI arm (mean 5.20, SD 0.61 vs 5.39, SD 0.66; adjusted &#x03B2;=&#x2212;0.199, 95% CI &#x2212;0.382 to &#x2212;0.016; <italic>P</italic>=.03). Usability/readability was unchanged, and cognitive load did not increase. Manipulation checks indicated higher cue recognition in the Safety UI arm.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>In a randomized static-vignette survey of older Chinese adults, a brief safety UI bundle was associated with higher verification intention and a trust calibration index consistent with lower overreliance risk, without detectable reductions in comprehension or usability/readability. Because the intervention was tested as a bundle using screenshots and generic source labels, findings should be interpreted as evidence for a practical interface-level strategy rather than proof that any single cue caused the observed effects.</p></sec></abstract><kwd-group><kwd>generative AI</kwd><kwd>large language models</kwd><kwd>user interface</kwd><kwd>safety cues</kwd><kwd>verification</kwd><kwd>trust calibration</kwd><kwd>older adults</kwd><kwd>China</kwd><kwd>vignette survey</kwd><kwd>randomized experiment</kwd><kwd>human-computer interaction</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Generative AI chat systems are rapidly becoming a default interface for information seeking and everyday decision support, offering users unprecedented access to synthesized knowledge and conversational assistance across domains such as health, education, and daily living [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. However, the systems are prone to producing plausible but incorrect or incomplete responses, so-called &#x201C;hallucinations,&#x201D; which can mislead users if not properly scrutinized [<xref ref-type="bibr" rid="ref2">2</xref>-<xref ref-type="bibr" rid="ref4">4</xref>]. The safe use of generative AI thus depends not only on the underlying model&#x2019;s quality but also on how user interfaces (UIs) shape users&#x2019; trust calibration and verification behaviors, especially in high-stakes contexts like digital health or personal decision-making [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>]. Within human-computer interaction (HCI) and digital health risk frameworks, it is increasingly recognized that interface design plays a critical role in guiding appropriate reliance on AI outputs without overclaiming clinical or societal consequences [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref6">6</xref>].</p><p>Older adults represent a particularly high-impact group for generative AI adoption. They stand to benefit substantially from accessible digital tools for health management, communication, and service navigation [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. Yet, older adults may also be more vulnerable to overreliance on automated systems due to lower digital literacy, age-related accessibility constraints, and distinct mental models of technology [<xref ref-type="bibr" rid="ref9">9</xref>-<xref ref-type="bibr" rid="ref11">11</xref>]. Research consistently finds that older users face barriers such as small font sizes, complex navigation, and unfamiliar interaction paradigms that can impede effective use or foster misplaced trust [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref12">12</xref>]. If UI design can nudge verification behaviors in older adults, encouraging them to check sources or question uncertain outputs, without sacrificing usability or increasing cognitive load, it offers a scalable lever for improving safety in real-world deployments [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>].</p><p>The concept of trust calibration is central to understanding safe human-AI interaction. Trust calibration refers to the alignment between a user&#x2019;s reliance on an automated system and the system&#x2019;s actual capabilities and uncertainties [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref4">4</xref>]. Classic literature in automation and HCI demonstrates that people can undertrust (ignoring helpful advice) or overtrust (accepting erroneous recommendations), with both extremes posing risks [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref6">6</xref>]. Interface cues, such as explanations, uncertainty indicators, source attributions, warnings, and verification nudges, are known to influence trust formation and reliance decisions [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref13">13</xref>]. Recent work in human-centered AI emphasizes the need for adaptive trust calibration mechanisms that help users match their level of scrutiny to the reliability of the system&#x2019;s output [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref6">6</xref>].</p><p>A growing body of research has explored various &#x201C;UI safety cues&#x201D; designed to improve trust calibration in AI and information systems. Such cues include providing explanations for outputs, communicating uncertainty levels, displaying source citations or references, issuing warnings or disclaimers about potential limitations, and designing friction into high-risk decisions [<xref ref-type="bibr" rid="ref14">14</xref>-<xref ref-type="bibr" rid="ref19">19</xref>].</p><p>Despite the advances, several gaps remain. Most empirical evidence comes from general adult samples in Western contexts or controlled laboratory tasks rather than real-world settings involving older Chinese adults, a population with unique language needs, information ecosystems, and user expectations [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref20">20</xref>]. Few studies have isolated simple UI bundles that are both effective at nudging verification intentions and feasible for deployment in commercial products without retraining underlying models [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>]. Moreover, much prior work relies heavily on self-reported measures of trust rather than behavioral proxies (eg, actual verification actions), limiting the strength of causal inference about interface effects on safety-related behaviors [<xref ref-type="bibr" rid="ref3">3</xref>].</p><p>This design-evidence gap is particularly salient for design and development teams seeking practical interventions. There is a need for UI changes that can be implemented quickly, without altering model internals, and evaluated efficiently through randomized experiments.</p><p>The China context further underscores the importance of targeted research. China has high smartphone penetration among older adults, widespread adoption of messaging platforms (eg, WeChat [Tencent Holdings Ltd]), increasing availability of AI-powered assistants across services, and a rapidly aging population facing unique digital inclusion challenges [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref20">20</xref>]. A China-specific sample enhances external validity by accounting for differences in language processing preferences, local information environments (including censorship or misinformation risks), and culturally shaped expectations about technology authority versus autonomy [<xref ref-type="bibr" rid="ref10">10</xref>].</p><p>This study addresses the gaps by testing whether adding a safety UI bundle&#x2014;a set of interface cues including uncertainty indicators and citation displays&#x2014;to a generative AI chat interface increases verification intention among older Chinese adults. The study further examines how this bundle affects related outcomes, including reliance intention (risk of overtrust), perceived trustworthiness of the system, comprehension of content, usability ratings, and subjective cognitive load. Importantly, this work focuses on UI-level interventions rather than model-level performance evaluation.</p><p>The anticipated contribution related to older adults was not that all older Chinese adults would necessarily respond in a uniform direction, but that safety cues would be evaluated under conditions common in later-life digital use, including variable digital literacy, high reliance on mobile interfaces, strong need for plain-language guidance, and possible dependence on family or staff assistance. Evidence from younger or more digitally experienced samples may not adequately capture whether low-burden cues remain noticeable, usable, and behaviorally meaningful in this population.</p><p>By providing randomized evidence on a practical bundle of interface cues tailored for older Chinese users and by quantifying the trade-off between increased verification orientation versus potential reductions in perceived trustworthiness, the study offers cautious design evidence for developers targeting safer AI use while recognizing that component-level effects require separate testing.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>We conducted this cross-sectional survey with an embedded randomized UI vignette experiment between May 22, 2025, and September 3, 2025. A 2-arm, between-subjects design compared a baseline generative AI chat interface with a safety UI bundle. Participants were randomized in a 1:1 ratio. Each participant viewed 2 scenarios in the assigned UI condition and completed outcome measures after each scenario. Analyses followed an intention-to-treat approach based on randomized assignment.</p></sec><sec id="s2-2"><title>Setting and Participant Recruitment</title><p>Participants were Chinese adults aged 60 years or older residing in mainland China. Recruitment was conducted through 3 channels, including community-based recruitment at senior activity centers or neighborhood community sites, outpatient clinic waiting area recruitment with on-site research staff, and online recruitment via WeChat groups that included older adults. To reduce exclusion of individuals with lower digital literacy, family-assisted or interviewer-assisted completion was permitted. Assistance mode was recorded for all participants and was included as a covariate and in sensitivity analyses. The survey was administered in simplified Chinese.</p></sec><sec id="s2-3"><title>Eligibility Criteria</title><p>Inclusion criteria were age 60 years or older, residence in mainland China, ability to read Chinese or complete the survey with assistance, and provision of informed consent. Exclusion criteria were inability to provide informed consent, duplicate submissions identified using platform controls plus response-pattern review, and low-quality responses defined a priori as failure on an instructed-response attention check combined with an implausibly short completion time.</p></sec><sec id="s2-4"><title>Duplicate Detection</title><p>For online recruitment, the survey platform settings were used to reduce repeat submissions, and additional screening was performed using completion time, highly similar response patterns, and device or IP-based indicators available to the platform. For community and clinic recruitment, staff recorded whether a participant had already completed the survey on the same day, and responses were additionally screened post hoc for near-identical response patterns and implausible repetition across entries.</p></sec><sec id="s2-5"><title>Randomization and Allocation Concealment</title><p>Randomization was implemented within the survey platform using built-in equal-probability assignment in a 1:1 ratio. Allocation was concealed until assignment. Condition assignment was maintained across both scenarios for each participant.</p></sec><sec id="s2-6"><title>Experimental Conditions and Rationale for UI Choices</title><p>The intervention was defined as a practical safety UI bundle rather than separable component-level manipulations. The study was not designed to attribute effects to individual elements within the bundle, and inferences were limited to the bundled condition compared with baseline.</p><p>The bundle development process was pragmatic and reproducibility-oriented. The research team first identified interface elements that could be implemented without model retraining or retrieval-system changes, then selected low-salience source labels and a short uncertainty plus verification message that could fit within a mobile chat layout. Candidate wording was simplified to avoid technical terms, preserve readability for older adults, and avoid giving the impression that the generic source labels represented verified citations.</p><sec id="s2-6-1"><title>Condition 1: Baseline UI</title><p>Participants viewed static screenshots of a mobile chat interface displaying one user prompt and one generative AI answer. The interface did not display additional safety cues under the answer. Screenshots of the baseline interface for both scenarios are provided in Figure S1A and S1C in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-6-2"><title>Condition 2: Safety UI Bundle</title><p>Participants viewed screenshots of the same interface and the same prompt and answer text, with 2 additional UI elements placed under the answer.</p><sec id="s2-6-2-1"><title>Source-Label Cue</title><p>A &#x201C;&#x6765;&#x6E90;&#x201D; (source) label followed by 2-3 neutral source chips, defined here as compact pill-shaped labels placed below the AI answer (for example &#x201C;&#x6743;&#x5A01;&#x6765;&#x6E90;A&#x201D; [authoritative source A], &#x201C;&#x6743;&#x5A01;&#x6765;&#x6E90;B&#x201D; [authoritative source B], and &#x201C;&#x516C;&#x5171;&#x670D;&#x52A1;&#x4FE1;&#x606F;&#x6765;&#x6E90;&#x201D; [public service information source]). To avoid implying verified provenance, the chips were intentionally generic and were presented as interface labels rather than citations of underlying model retrieval. A brief note indicated that source information could be expanded.</p></sec><sec id="s2-6-2-2"><title>Uncertainty and Verification Nudge</title><p>A short panel stated that the answer might be incomplete or not suitable for individual circumstances and recommended checking authoritative sources and consulting professionals when appropriate. The wording was drafted by the research team with attention to plain Chinese, low literacy burden, and avoidance of diagnostic or individualized clinical instruction; however, it was not developed through a separate formal linguistic validation process or review by a behavioral health expert panel.</p></sec></sec></sec><sec id="s2-7"><title>Vignette Scenarios and Stimulus Construction</title><p>Two scenarios were developed to reflect common information needs among older adults while minimizing potential ethical and clinical risks. Scenario 1 addressed an everyday service or travel decision query. Scenario 2 addressed general well-being related to sleep quality and daytime fatigue. The generative AI answers were written conservatively and avoided diagnosis, medication dosing, or individualized clinical instruction.</p><p>Stimuli were presented as static mobile UI screenshots with large, high-contrast text. The interface layout, typography, spacing, icons, and answer length were held constant across conditions. Only the safety UI elements differed between arms. Participants could not interact with the chat interface itself, so the experiment captured responses to a controlled representation of an AI chat interface rather than the cognitive demands of real-time conversational use. Screenshots of the safety UI bundle for both scenarios are provided in Figures S1A-S2B and in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-8"><title>Scenario Order</title><p>To reduce order effects, the presentation order of the 2 scenarios was randomized at the participant level by the survey platform. Participants remained in the same assigned UI condition regardless of scenario order.</p></sec><sec id="s2-9"><title>Procedure</title><p>After informed consent and eligibility screening, participants completed baseline measures including demographics, digital literacy, technology anxiety, and prior exposure to AI chat tools. Participants were then randomized to the baseline UI or safety UI bundle condition. Participants viewed the first scenario screenshot (assigned condition) and completed scenario-specific outcome measures, a brief manipulation check, and a comprehension quiz. Participants then viewed the second scenario screenshot (same assigned condition) and completed the same measures. The survey concluded with brief global attitude items and an optional open-ended question about desired interface features. Completion time and per-page dwell time were recorded automatically. Assistance mode (independent, family-assisted, and interviewer-assisted) was recorded for all participants.</p><p>Family-assisted or interviewer-assisted completion was allowed when participants had visual, motor, or digital-literacy barriers. During assisted completion, participants were shown the assigned screenshots whenever possible. If reading support was required, assistants were instructed to read the visible on-screen text and survey questions verbatim, including any source-label text or uncertainty/verification nudge displayed in the assigned condition, without explaining, emphasizing, or interpreting the interface cues or the intended purpose of the intervention. Assistance mode was recorded for all participants.</p></sec><sec id="s2-10"><title>Behavioral Proxy Measure of Verification Behavior</title><p>To strengthen outcome validity beyond self-report, a behavioral proxy for verification was included after each scenario. Participants were shown a small, nonemphasized expandable element labeled &#x201C;&#x6765;&#x6E90;&#x4FE1;&#x606F; (&#x53EF;&#x5C55;&#x5F00;)&#x201D; (source information [expandable]) placed below the outcome questions, not directly under the screenshot. Expanding it opened a brief neutral source information panel. The outcome was whether the participant expanded it at least once. This measure was interpreted as a low-friction proxy for curiosity about source information, not as evidence of substantive real-world verification such as cross-checking external websites, consulting clinicians, or comparing multiple information sources.</p></sec><sec id="s2-11"><title>Measures</title><p>All Likert items were measured on a 7-point scale from &#x201C;&#x975E;&#x5E38;&#x4E0D;&#x540C;&#x610F;&#x201D; (strongly disagree) to &#x201C;&#x975E;&#x5E38;&#x540C;&#x610F;&#x201D; (strongly agree), unless otherwise specified.</p></sec><sec id="s2-12"><title>Manipulation Checks</title><p>After each scenario, participants completed two brief manipulation check items: (1) a yes/no item assessing whether the interface displayed a source information prompt, and (2) a Likert item assessing whether they noticed &#x201C;&#x63D0;&#x793A;&#x7C7B;&#x4FE1;&#x606F;&#x201D; (tip-style information, including reminders to verify or consult professionals).</p></sec><sec id="s2-13"><title>Primary Outcomes</title><p>Verification intention was the primary outcome of the study. For each scenario, verification intention was measured using a 3-item study-created scale informed by prior work on information verification and AI reliance. The items assessed intention to cross-check information via other channels, preference for authoritative sources, and intention to consult professionals or experienced individuals before acting. Scores were averaged across scenarios and items; higher scores indicated stronger verification intention.</p><p>Reliance intention was a co-primary outcome. For each scenario, reliance intention was measured using a 3-item study-created scale informed by automation-trust and AI-adoption literature. The items assessed willingness to follow the advice, perceived sufficiency to adopt the advice directly, and likelihood of using the AI chat interface again for similar problems. A reliance intention score was averaged across scenarios and items; higher scores indicated stronger direct reliance intention.</p><p>The behavioral proxy expansion outcome was treated as a co-primary supportive measure. It was defined as expanding &#x201C;&#x6765;&#x6E90;&#x4FE1;&#x606F; (&#x53EF;&#x5C55;&#x5F00;)&#x201D; at least once, with scenario-level expansion reported as a secondary breakdown.</p></sec><sec id="s2-14"><title>Secondary Outcomes</title><sec id="s2-14-1"><title>Trust Calibration Index</title><p>A trust calibration index was computed as an exploratory secondary representation of calibration. It was defined a priori as the standardized difference between reliance intention and verification intention, averaged across scenarios, with higher values indicating greater overreliance risk. Reliance and verification intention were standardized within the analytic sample before subtraction. This index was study-created and was not a previously validated psychometric instrument; it was intended to operationalize the balance between direct reliance and verification orientation in this vignette context.</p></sec><sec id="s2-14-2"><title>Comprehension</title><p>Four study-created multiple-choice questions per scenario assessed understanding of key points stated in the answer, including the main recommendation, stated uncertainty or caveat, appropriate next action, and an item requiring recognition of information that the answer did not provide. Each item was scored as correct or incorrect, yielding a total comprehension score from 0 to 8 across scenarios.</p></sec><sec id="s2-14-3"><title>Perceived Trustworthiness</title><p>A 4-item study-created composite, informed by human-AI trust literature, assessed perceived competence, overall reliability, perceived risk of being misleading (reverse-coded), and perceived helpfulness. Scores were averaged per scenario and across scenarios. Internal consistency was assessed in the analytic sample.</p></sec><sec id="s2-14-4"><title>Perceived Usability and Readability</title><p>A 5-item study-created composite assessed text readability, ease of locating key points, ease of understanding, confidence in using the interface, and overall satisfaction. The composite was not a direct administration of the System Usability Scale because the experiment used static screenshots rather than a fully interactive system. Scores were averaged per scenario and across scenarios, and internal consistency was assessed.</p></sec><sec id="s2-14-5"><title>Cognitive Load</title><p>A single mental-effort item per scenario, adapted from established subjective cognitive load measurement practice, was rated from 0 (no effort) to 10 (very high effort) and assessed perceived effort to read and understand the answer. Scores were averaged across scenarios.</p></sec></sec><sec id="s2-15"><title>Determinants and Covariates</title><p>Digital literacy was measured using a 6-item study-created screener assessing self-rated ability to perform smartphone-based information search, app installation and updating, font-size adjustment, voice input use, privacy settings management, and handling common login or verification steps. Items were scored on a 5-point scale and averaged, with higher scores indicating greater digital literacy. Participants who completed the survey with family or interviewer assistance were not excluded from digital literacy analyses solely because assistance was provided. Their digital literacy data were retained as participant-reported baseline characteristics. Assistance mode was recorded separately for all participants and was included as a covariate in adjusted models to account for potential differences between independent, family-assisted, and interviewer-assisted completion.</p><p>Technology anxiety was measured using a 4-item study-created scale informed by prior technology-anxiety constructs. The items assessed nervousness with new technology, worry about making mistakes, avoidance of unfamiliar features, and perceived stress when learning technology. Items were averaged, with higher scores indicating greater technology anxiety.</p><p>Covariates included age, gender, education level, residence type, living arrangement, self-rated health, prior AI chat exposure, and assistance mode.</p></sec><sec id="s2-16"><title>Data Quality Procedures</title><p>The survey included an instructed-response attention check. Duplicate prevention used survey platform settings plus post hoc response-pattern checks. Dwell time and completion time were recorded and used as quality indicators rather than stand-alone enforced thresholds to avoid disproportionate burden on older adults who may need more time or assistance. Low-quality responding was defined before analysis as failure of the instructed-response attention check plus either an implausibly short total completion time (&#x003C;3 minutes) or a straight-line response pattern across all Likert outcome items.</p></sec><sec id="s2-17"><title>Sample Size</title><p>The target sample size was set at 200 participants (approximately 100 per arm). This sample size was selected to support estimation of the between-group difference in the primary continuous outcome (verification intention) with adequate precision while remaining feasible across the recruitment channels. Using a 2-sample comparison framework with equal allocation, 2-sided &#x03B1; of .05, and a continuous outcome, a total sample of 200 provided approximately 80% power to detect a standardized mean difference of about 0.4 between arms. The sample size also supported adjusted regression models including prespecified covariates without unstable estimation. The study was powered for the primary main effect comparison; heterogeneity analyses were interpreted descriptively.</p></sec><sec id="s2-18"><title>Statistical Analysis</title><p>All analyses were conducted under intention-to-treat based on randomized assignment. Baseline balance between arms was described using standardized mean differences for key demographic and technology-related variables.</p><p>Primary analyses estimated the effect of condition (safety UI bundle vs baseline) on verification intention and reliance intention using linear regression with robust SEs, adjusting for prespecified covariates (age, gender, education, prior AI chat exposure, digital literacy, technology anxiety, and assistance mode). Results were reported as adjusted mean differences with 95% CIs. For verification intention, an unadjusted Cohen <italic>d</italic> was also reported descriptively.</p><p>The behavioral proxy expansion outcome was analyzed using logistic regression estimating the odds of expanding &#x201C;&#x6765;&#x6E90;&#x4FE1;&#x606F; (&#x53EF;&#x5C55;&#x5F00;)&#x201D; at least once, adjusting for the same covariates. Scenario-level expansion outcomes were summarized descriptively and analyzed in secondary models.</p><p>Secondary outcomes, including trust calibration index, perceived trustworthiness, usability/readability, and cognitive load, were analyzed using analogous regression models. For the comprehension score, distributional diagnostics were examined to determine whether a count model was necessary; because diagnostics did not indicate material model misspecification, the final analysis used linear regression with robust SEs, and effects were reported as adjusted mean differences on the 0-8 score scale.</p><p>Because assistance during completion could influence participants&#x2019; ability to complete the survey and could be associated with digital literacy, assistance mode was adjusted for in the primary models. In addition, a sensitivity analysis restricted to participants who completed the survey independently was conducted to assess whether the main findings were robust to exclusion of assisted completions.</p><p>A condition-by-digital literacy interaction term was examined to assess whether the intervention effect varied with digital literacy, and results were interpreted alongside uncertainty intervals.</p></sec><sec id="s2-19"><title>Missing Data Handling</title><p>Item-level missingness was below the predefined 5% threshold for all primary and secondary outcomes; therefore, complete-case analysis was used for the final analysis. Multiple imputation by chained equations was not applied.</p></sec><sec id="s2-20"><title>Ethical Considerations</title><p>Ethics approval was obtained from the Ethics Committee of Chengdu Gongyun Education and Management Research Institute (approval number GYLL25013). Although this study uses a randomized allocation design to present different user interface variants, it is classified as a cross-sectional vignette survey rather than a clinical randomized controlled trial involving a live intervention, and therefore clinical trial database registration was not required. Because the study involved survey responses and static vignettes rather than clinical intervention, it was classified as minimal risk. All participants provided informed consent before eligibility screening and randomization. Study data were deidentified before analysis, and no identifying clinical information was collected. No participant images or identifiable user screenshots were included in the manuscript or supplementary materials. Participants did not receive monetary compensation. The well-being scenario provided general nondiagnostic information only and advised professional consultation when appropriate. Family-assisted or interviewer-assisted completion was permitted for accessibility; research staff were not involved in outcome interpretation, but on-site staff were not formally blinded to randomized condition after assignment because screenshots were visible during assisted completion.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Internal Consistency and Descriptive Performance of Study-Created Measures</title><p>Item-level missingness was 0% for the study-created multi-item measures used in the primary and covariate-adjusted analyses. Internal consistency was good to excellent for the 2 primary self-report outcomes. Cronbach &#x03B1; was 0.890 for the 6-item verification intention score and 0.873 for the 6-item reliance intention score. Scenario-specific coefficients were also acceptable for verification intention and reliance intention, as shown in <xref ref-type="table" rid="table1">Table 1</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Internal consistency and scoring of study-created multi-item measures.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Measure<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="bottom">Items</td><td align="left" valign="bottom">Range</td><td align="left" valign="bottom">Score construction</td><td align="left" valign="bottom">Cronbach &#x03B1;</td><td align="left" valign="bottom">Scenario 1/2 (Cronbach &#x03B1;)</td><td align="left" valign="bottom">Overall mean (SD)</td></tr></thead><tbody><tr><td align="left" valign="top">Verification intention</td><td align="left" valign="top">6</td><td align="left" valign="top">1&#x2010;7</td><td align="left" valign="top">Mean of 3 items in each of 2 scenarios</td><td align="left" valign="top">0.890</td><td align="left" valign="top">0.829/0.785</td><td align="left" valign="top">4.57 (0.63)</td></tr><tr><td align="left" valign="top">Reliance intention</td><td align="left" valign="top">6</td><td align="left" valign="top">1&#x2010;7</td><td align="left" valign="top">Mean of 3 items in each of 2 scenarios</td><td align="left" valign="top">0.873</td><td align="left" valign="top">0.770/0.792</td><td align="left" valign="top">5.00 (0.56)</td></tr><tr><td align="left" valign="top">Digital literacy</td><td align="left" valign="top">6</td><td align="left" valign="top">1&#x2010;5</td><td align="left" valign="top">Mean of 6 baseline screener items</td><td align="left" valign="top">0.959</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">3.14 (0.78)</td></tr><tr><td align="left" valign="top">Technology anxiety</td><td align="left" valign="top">4</td><td align="left" valign="top">1&#x2010;7</td><td align="left" valign="top">Mean of 4 baseline anxiety items</td><td align="left" valign="top">0.941</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">3.66 (1.27)</td></tr><tr><td align="left" valign="top">Perceived trustworthiness</td><td align="left" valign="top">8</td><td align="left" valign="top">1&#x2010;7</td><td align="left" valign="top">Mean of 4 items in each of 2 scenarios</td><td align="left" valign="top">0.930</td><td align="left" valign="top">0.855/0.881</td><td align="left" valign="top">5.29 (0.66)</td></tr><tr><td align="left" valign="top">Usability/readability</td><td align="left" valign="top">10</td><td align="left" valign="top">1&#x2010;7</td><td align="left" valign="top">Mean of 5 items in each of 2 scenarios</td><td align="left" valign="top">0.946</td><td align="left" valign="top">0.898/0.901</td><td align="left" valign="top">5.40 (0.63)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Verification intention, reliance intention, perceived trustworthiness, and usability/readability were measured after each of the 2 vignette scenarios and then averaged at the participant level. Digital literacy and technology anxiety were baseline covariate scales.</p></fn><fn id="table1fn2"><p><sup>b</sup>Not applicable.</p></fn></table-wrap-foot></table-wrap><p>The study-created covariate scales also showed high internal consistency. Cronbach &#x03B1; was 0.959 for the 6-item digital literacy screener and 0.941 for the 4-item technology anxiety scale. For the secondary composite outcomes, &#x03B1; was 0.930 for perceived trustworthiness and 0.946 for usability/readability. The scoring rules, overall descriptive statistics, and reliability coefficients for the study-created multi-item measures are summarized in <xref ref-type="table" rid="table1">Table 1</xref>.</p><p>The descriptive distributions were consistent with the primary outcome pattern and did not indicate problematic floor or ceiling effects for the main self-report scales. Verification intention was higher in the Safety UI bundle arm, whereas reliance intention was similar between arms. Digital literacy and technology anxiety were also similar between randomized arms, supporting their role as covariates rather than major sources of baseline imbalance. Descriptive performance for the study-created scales and nonscale measures is reported in <xref ref-type="table" rid="table2">Table 2</xref>.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Descriptive performance of study-created measures by randomized arm.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Measure</td><td align="left" valign="bottom">Range</td><td align="left" valign="bottom">Baseline UI<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> (n=100)</td><td align="left" valign="bottom">Safety UI bundle (n=100)</td><td align="left" valign="bottom">Interpretation</td></tr></thead><tbody><tr><td align="left" valign="top">Verification intention, mean (SD)</td><td align="left" valign="top">1&#x2010;7</td><td align="left" valign="top">4.41 (0.59)</td><td align="left" valign="top">4.72 (0.63)</td><td align="left" valign="top">Higher scores indicate stronger intention to verify.</td></tr><tr><td align="left" valign="top">Reliance intention, mean (SD)</td><td align="left" valign="top">1&#x2010;7</td><td align="left" valign="top">5.03 (0.58)</td><td align="left" valign="top">4.97 (0.54)</td><td align="left" valign="top">Higher scores indicate stronger direct reliance.</td></tr><tr><td align="left" valign="top">Digital literacy, mean (SD)</td><td align="left" valign="top">1&#x2010;5</td><td align="left" valign="top">3.12 (0.78)</td><td align="left" valign="top">3.15 (0.78)</td><td align="left" valign="top">Baseline covariate; higher scores indicate greater digital literacy.</td></tr><tr><td align="left" valign="top">Technology anxiety, mean (SD)</td><td align="left" valign="top">1&#x2010;7</td><td align="left" valign="top">3.66 (1.29)</td><td align="left" valign="top">3.67 (1.25)</td><td align="left" valign="top">Baseline covariate; higher scores indicate greater technology anxiety.</td></tr><tr><td align="left" valign="top">Perceived trustworthiness, mean (SD)</td><td align="left" valign="top">1&#x2010;7</td><td align="left" valign="top">5.39 (0.70)</td><td align="left" valign="top">5.20 (0.61)</td><td align="left" valign="top">Higher scores indicate greater perceived trustworthiness.</td></tr><tr><td align="left" valign="top">Usability/readability, mean (SD)</td><td align="left" valign="top">1&#x2010;7</td><td align="left" valign="top">5.40 (0.60)</td><td align="left" valign="top">5.39 (0.66)</td><td align="left" valign="top">Higher scores indicate better usability/readability.</td></tr><tr><td align="left" valign="top">Comprehension total, mean (SD)</td><td align="left" valign="top">0&#x2010;8</td><td align="left" valign="top">6.32 (1.08)</td><td align="left" valign="top">6.33 (1.14)</td><td align="left" valign="top">Objective quiz total; alpha was not interpreted.</td></tr><tr><td align="left" valign="top">Cognitive load, mean (SD)</td><td align="left" valign="top">0&#x2010;10</td><td align="left" valign="top">3.96 (1.60)</td><td align="left" valign="top">3.56 (1.44)</td><td align="left" valign="top">Mean of one mental-effort item per scenario.</td></tr><tr><td align="left" valign="top">Expanded source information at least once, n (%)</td><td align="left" valign="top">0/1</td><td align="left" valign="top">27 (27.0)</td><td align="left" valign="top">42 (42.0)</td><td align="left" valign="top">Behavioral proxy; not a psychometric scale.</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>UI: user interface.</p></fn></table-wrap-foot></table-wrap><p>The comprehension outcome was scored as an 8-item objective quiz total and was not treated as a reflective psychometric scale because the items intentionally sampled distinct factual and caveat-related elements from the 2 vignettes. Cognitive load was assessed using one mental-effort rating per scenario, and source-information expansion was a binary behavioral proxy; therefore, Cronbach &#x03B1; was not interpreted for these measures.</p><p>Overall, the reliability and descriptive results support the internal consistency and scoring adequacy of the study-created verification intention, reliance intention, digital literacy, technology anxiety, perceived trustworthiness, and usability/readability scores. These findings provide measurement support for the adjusted intention-to-treat analyses of the primary and secondary outcomes.</p></sec><sec id="s3-2"><title>Participant Flow and Analytic Sample</title><p>A total of 236 individuals were approached across the 3 recruitment channels. Twenty-two individuals did not consent or did not start the survey. Of 214 who provided consent and started the survey, 14 were excluded before analysis because of ineligibility (n=6, &#x003C;60 years), duplicate or near-duplicate entries (n=4), or low-quality responding defined as failing the attention check with an implausibly short completion time (n=4). The final analytic sample included 200 participants, with 100 randomized to the Baseline UI arm and 100 randomized to the Safety UI bundle arm (<xref ref-type="fig" rid="figure1">Figure 1</xref>).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Participant flow diagram for a randomized vignette survey of a safety user interface (UI) bundle for generative AI chat use among older Chinese adults in mainland China, May 22, 2025-September 3, 2025. The diagram shows the numbers approached, not consenting or not starting the survey, consenting and starting the survey, excluded before analysis, and included in each randomized arm, with reasons for exclusion.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e94140_fig01.png"/></fig></sec><sec id="s3-3"><title>Baseline Characteristics</title><p>Randomization produced broadly comparable arms (<xref ref-type="table" rid="table3">Table 3</xref>), although some baseline differences were observed. The mean age was 67.48 (SD 5.14) years in the Baseline UI arm and 67.65 (SD 5.18) years in the Safety UI arm. Female participants accounted for 54.0% (n=54) in the Baseline UI arm and 44.0% (n=44) in the Safety UI arm. College or higher education was reported by 14.0% (n=14) in the Baseline UI arm and 30.0% (n=30) in the Safety UI bundle arm. Prior use of AI chat tools was reported by 40.0% (n=40) in the Baseline UI arm and 41.0% (n=41) in the Safety UI arm. Standardized mean differences were largest for education and gender, so adjusted analyses retained prespecified covariates to reduce residual confounding.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Baseline demographic, health, technology-use, and survey-completion characteristics of participants (N=200).</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Characteristic</td><td align="left" valign="bottom">Baseline UI<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup> (n=100)</td><td align="left" valign="bottom">Safety UI bundle (n=100)</td></tr></thead><tbody><tr><td align="left" valign="top">Age (years), mean (SD)</td><td align="left" valign="top">67.48 (5.14)</td><td align="left" valign="top">67.65 (5.18)</td></tr><tr><td align="left" valign="top">Sex (female), n (%)</td><td align="left" valign="top">54 (54.0)</td><td align="left" valign="top">44 (44.0)</td></tr><tr><td align="left" valign="top">Education, n (%)</td><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Primary or less</td><td align="left" valign="top">14 (14.0)</td><td align="left" valign="top">16 (16.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Middle school</td><td align="left" valign="top">42 (42.0)</td><td align="left" valign="top">24 (24.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>High school or vocational</td><td align="left" valign="top">30 (30.0)</td><td align="left" valign="top">30 (30.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>College or higher</td><td align="left" valign="top">14 (14.0)</td><td align="left" valign="top">30 (30.0)</td></tr><tr><td align="left" valign="top">Residence type, n (%)</td><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Urban</td><td align="left" valign="top">50 (50.0)</td><td align="left" valign="top">47 (47.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>County</td><td align="left" valign="top">31 (31.0)</td><td align="left" valign="top">32 (32.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Rural</td><td align="left" valign="top">19 (19.0)</td><td align="left" valign="top">21 (21.0)</td></tr><tr><td align="left" valign="top">Assistance mode, n (%)</td><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Independent</td><td align="left" valign="top">68 (68.0)</td><td align="left" valign="top">66 (66.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Family-assisted</td><td align="left" valign="top">17 (17.0)</td><td align="left" valign="top">18 (18.0)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Interviewer-assisted</td><td align="left" valign="top">15 (15.0)</td><td align="left" valign="top">16 (16.0)</td></tr><tr><td align="left" valign="top">AI chat ever used, n (%)</td><td align="left" valign="top">40 (40.0)</td><td align="left" valign="top">41 (41.0)</td></tr><tr><td align="left" valign="top">Digital literacy (1-5), mean (SD)</td><td align="left" valign="top">3.12 (0.78)</td><td align="left" valign="top">3.15 (0.78)</td></tr><tr><td align="left" valign="top">Technology anxiety (1-7), mean (SD)</td><td align="left" valign="top">3.66 (1.29)</td><td align="left" valign="top">3.67 (1.25)</td></tr><tr><td align="left" valign="top">Self-rated health (1-5), mean (SD)</td><td align="left" valign="top">3.46 (0.82)</td><td align="left" valign="top">3.46 (0.90)</td></tr><tr><td align="left" valign="top">Scenario 1 presented first, n (%)</td><td align="left" valign="top">52 (52.0)</td><td align="left" valign="top">52 (52.0)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>UI: user interface.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-4"><title>Manipulation Checks</title><p>The manipulation checks confirmed that participants perceived the intended UI differences. In the Safety UI bundle arm, 75.0% (n=75) reported noticing a source prompt, compared with 15.0% (n=15) in the Baseline UI arm. The &#x201C;noticed tip&#x201D; item also differed in the expected direction, with mean 5.20 (SD 1.13) in the Safety UI bundle arm and mean 3.26 (SD 1.15) in the Baseline UI arm (<xref ref-type="table" rid="table4">Table 4</xref>).</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Manipulation checks after exposure to baseline and safety user interface (UI) bundle screenshots (N=200).</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Measure</td><td align="left" valign="bottom">Baseline UI<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup> (n=100)</td><td align="left" valign="bottom">Safety UI bundle (n=100)</td></tr></thead><tbody><tr><td align="left" valign="top">Noticed source prompt, n (%)</td><td align="left" valign="top">15 (15.0)</td><td align="left" valign="top">75 (75.0)</td></tr><tr><td align="left" valign="top">Noticed tip item (1-7), mean (SD)</td><td align="left" valign="top">3.26 (1.15)</td><td align="left" valign="top">5.20 (1.13)</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>UI: user interface.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-5"><title>Primary Outcomes</title><p>Verification intention was higher in the Safety UI bundle arm than in the Baseline UI arm. The mean verification intention score was 4.41 (SD 0.59) in the Baseline UI arm and 4.72 (SD 0.63) in the Safety UI bundle arm, corresponding to an unadjusted mean difference of 0.31 points on the 7-point scale (Cohen <italic>d</italic>=0.50). In covariate-adjusted regression, assignment to the Safety UI bundle was associated with a 0.293 point increase in verification intention (95% CI 0.128-0.457; <italic>P</italic>&#x003C;.001), holding constant age, gender, education, prior AI chat use, digital literacy, technology anxiety, and assistance mode.</p><p>Reliance intention was similar between arms. The mean reliance intention score was 5.03 (SD 0.58) in the Baseline UI arm and 4.97 (SD 0.54) in the Safety UI bundle arm, with a small unadjusted difference of &#x2212;0.06. In covariate-adjusted regression, the estimated association between the Safety UI bundle and reliance intention was &#x2212;0.105 (95% CI &#x2212;0.239 to 0.029; <italic>P</italic>=.13).</p><p>The behavioral proxy outcome showed a numerically higher proportion of participants expanding the optional &#x201C;source information&#x201D; element in the Safety UI bundle arm, but the adjusted estimate was not statistically definitive. Expansion at least once occurred in 27.0% (n=27) of Baseline UI participants and 42.0% (n=42) of Safety UI participants. In adjusted logistic regression, the odds ratio (OR) for expansion was 1.76 (95% CI 0.947-3.269; <italic>P</italic>=.07), suggesting a possible increase in source-information seeking, although the CI included the null.</p><p>Scenario-level expansion proportions were numerically higher in the Safety UI bundle arm, but these secondary binary outcomes were interpreted cautiously. In Scenario 1, expansion occurred in 21.0% (n=21) of participants in the Baseline UI arm and 34.0% (n=34) in the Safety UI bundle arm. In the adjusted logistic regression model controlling for age, gender, education, prior AI chat exposure, digital literacy, technology anxiety, and assistance mode, assignment to the Safety UI bundle was associated with higher odds of Scenario 1 expansion, although the CI included the null (adjusted OR=1.86, 95% CI 0.94-3.68; <italic>P</italic>=.08). In Scenario 2, expansion occurred in 18.0% (n=18) of participants in the Baseline UI arm and 31.0% (n=31) in the Safety UI bundle arm. The adjusted scenario-level model similarly favored the Safety UI bundle, again with uncertainty around the estimate (adjusted OR=1.92, 95% CI 0.95-3.88; <italic>P</italic>=.07). These scenario-level models were interpreted as secondary supportive analyses because the study was powered for the main continuous outcome rather than for scenario-specific binary expansion outcomes. Global attitude items were exploratory and showed no evidence that the safety UI bundle reduced willingness to use AI chat tools for low-risk everyday information seeking. Optional open-ended responses were sparse and were not treated as formal outcomes (<xref ref-type="table" rid="table5">Table 5</xref>).</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Primary verification, reliance, and behavioral proxy outcomes by randomized arm (N=200).</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Outcome</td><td align="left" valign="bottom">Baseline UI<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup> (n=100)</td><td align="left" valign="bottom">Safety UI bundle (n=100)</td><td align="left" valign="bottom">Unadjusted difference<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup></td><td align="left" valign="bottom">Adjusted effect estimate<sup><xref ref-type="table-fn" rid="table5fn3">c</xref></sup>, estimate (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Verification intention (1-7), mean (SD)</td><td align="left" valign="top">4.41 (0.59)</td><td align="left" valign="top">4.72 (0.63)</td><td align="left" valign="top">0.31</td><td align="left" valign="top">&#x03B2;=0.293 (0.128 to 0.457)</td><td align="char" char="." valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Reliance intention (1-7), mean (SD)</td><td align="left" valign="top">5.03 (0.58)</td><td align="left" valign="top">4.97 (0.54)</td><td align="left" valign="top">&#x2013;0.06</td><td align="left" valign="top">&#x03B2;=&#x2013;0.105 (&#x2013;0.239 to 0.029)</td><td align="char" char="." valign="top">.13</td></tr><tr><td align="left" valign="top">Expanded source info at least once, n (%)</td><td align="left" valign="top">27 (27.0)</td><td align="left" valign="top">42 (42.0)</td><td align="left" valign="top">15.0</td><td align="left" valign="top">OR<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup> 1.76 (0.95 to 3.27)</td><td align="char" char="." valign="top">.07</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>UI: user interface.</p></fn><fn id="table5fn2"><p><sup>b</sup>Unadjusted difference was calculated as the Safety UI bundle arm minus the Baseline UI arm: mean difference for continuous outcomes and percentage point difference for the binary behavioral proxy.</p></fn><fn id="table5fn3"><p><sup>c</sup>Adjusted estimates controlled for age, gender, education, prior AI chat exposure, digital literacy, technology anxiety, and assistance mode.</p></fn><fn id="table5fn4"><p><sup>d</sup>OR: odds ratio.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-6"><title>Secondary Outcomes</title><p>Trust calibration, operationalized as the standardized difference between reliance and verification, favored the Safety UI bundle arm. The trust calibration index mean was 0.29 (SD 1.29) in the Baseline UI arm and &#x2212;0.29 (SD 1.43) in the Safety UI bundle arm, indicating lower overreliance risk when the safety UI bundle was present. The adjusted association also favored the Safety UI bundle (&#x03B2;=&#x2212;0.567, 95% CI &#x2212;1.005 to &#x2212;0.129; <italic>P</italic>=.01). Because the generative AI answers were intentionally conservative and did not include incorrect or hallucinated responses, the index should be interpreted as an exploratory balance between reliance and verification orientation rather than a definitive measure of correct trust calibration.</p><p>Comprehension scores were high in both arms and nearly identical. The mean comprehension total score was 6.32 (SD 1.08) in the Baseline UI arm and 6.33 (SD 1.14) in the Safety UI bundle arm. Distributional diagnostics did not indicate a need for a count model; therefore, the final adjusted analysis used linear regression with robust SEs. The adjusted association between arm and comprehension was small and not statistically distinguishable from zero (&#x03B2;=&#x2212;0.132, 95% CI &#x2212;0.428 to 0.163; <italic>P</italic>=.38).</p><p>Before outcome modeling, internal consistency was assessed for the 2 study-created multi-item secondary outcome composites. Cronbach &#x03B1; was 0.930 for perceived trustworthiness and 0.946 for usability/readability, supporting use of the averaged composite scores in the regression analyses.</p><p>Perceived trustworthiness showed a modest reduction in the Safety UI bundle arm. The mean perceived trustworthiness score was 5.39 (SD 0.70) in the Baseline UI arm and 5.20 (SD 0.61) in the Safety UI bundle arm. The adjusted &#x03B2; was &#x2212;0.199 (95% CI &#x2212;0.382 to &#x2212;0.016; <italic>P</italic>=.03), indicating that adding safety cues improved verification orientation while slightly lowering perceived trustworthiness.</p><p>Usability/readability ratings were similar between arms, with mean 5.40 (SD 0.60) in the Baseline UI arm and 5.39 (SD 0.66) in the Safety UI bundle arm. The adjusted association was small and not statistically significant (&#x03B2;=&#x2212;0.069, 95% CI &#x2212;0.254 to 0.115; <italic>P</italic>=.46). Cognitive load tended to be lower in the Safety UI bundle arm, with mean 3.96 (SD 1.60) in the Baseline UI arm and 3.56 (SD 1.44) in the Safety UI bundle arm, although the adjusted estimate did not exclude zero (&#x03B2;=&#x2212;0.346, 95% CI &#x2212;0.762 to 0.071; <italic>P</italic>=.10; <xref ref-type="fig" rid="figure2">Figure 2</xref> and <xref ref-type="table" rid="table6">Table 6</xref>).</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Trade-off between verification intention and perceived trustworthiness by randomized arm in a static generative AI chat vignette survey among older Chinese adults in mainland China, May 22, 2025-September 3, 2025. The scatter plot displays participant-level verification intention and perceived trustworthiness scores, with color indicating the randomized arm. UI: user interface.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e94140_fig02.png"/></fig><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Secondary trust calibration, comprehension, perceived trustworthiness, usability/readability, and cognitive load outcomes by randomized arm (N=200).</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Outcome</td><td align="left" valign="bottom">Baseline UI<sup><xref ref-type="table-fn" rid="table6fn1">a</xref></sup> (n=100)</td><td align="left" valign="bottom">Safety UI bundle (n=100)</td><td align="left" valign="bottom">Unadjusted difference<sup><xref ref-type="table-fn" rid="table6fn2">b</xref></sup></td><td align="left" valign="bottom">Adjusted effect estimate<sup><xref ref-type="table-fn" rid="table6fn3">c</xref></sup>, &#x03B2; (95% CI)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Trust calibration index, mean (SD)</td><td align="left" valign="top">0.29 (1.29)</td><td align="left" valign="top">&#x2013;0.29 (1.43)</td><td align="left" valign="top">&#x2013;0.58</td><td align="left" valign="top">&#x2013;0.567 (&#x2013;1.005 to &#x2212;0.129)</td><td align="char" char="." valign="top">.01</td></tr><tr><td align="left" valign="top">Comprehension total (0-8), mean (SD)</td><td align="left" valign="top">6.32 (1.08)</td><td align="left" valign="top">6.33 (1.14)</td><td align="left" valign="top">0.01</td><td align="left" valign="top">&#x2013;0.132 (&#x2013;0.428 to 0.163)</td><td align="char" char="." valign="top">.38</td></tr><tr><td align="left" valign="top">Perceived trustworthiness (1-7), mean (SD)</td><td align="left" valign="top">5.39 (0.70)</td><td align="left" valign="top">5.20 (0.61)</td><td align="left" valign="top">&#x2013;0.19</td><td align="left" valign="top">&#x2013;0.199 (&#x2013;0.382 to &#x2212;0.016)</td><td align="char" char="." valign="top">.03</td></tr><tr><td align="left" valign="top">Usability/readability (1-7), mean (SD)</td><td align="left" valign="top">5.40 (0.60)</td><td align="left" valign="top">5.39 (0.66)</td><td align="left" valign="top">&#x2013;0.01</td><td align="left" valign="top">&#x2013;0.069 (&#x2013;0.254 to 0.115)</td><td align="char" char="." valign="top">.46</td></tr><tr><td align="left" valign="top">Cognitive load (0-10), mean (SD)</td><td align="left" valign="top">3.96 (1.60)</td><td align="left" valign="top">3.56 (1.44)</td><td align="left" valign="top">&#x2013;0.40</td><td align="left" valign="top">&#x2013;0.346 (&#x2013;0.762 to 0.071)</td><td align="char" char="." valign="top">.10</td></tr></tbody></table><table-wrap-foot><fn id="table6fn1"><p><sup>a</sup>UI: user interface.</p></fn><fn id="table6fn2"><p><sup>b</sup>Unadjusted difference was calculated as the Safety UI bundle arm mean minus the Baseline UI arm mean, in the original units of each outcome.</p></fn><fn id="table6fn3"><p><sup>c</sup>Adjusted estimates controlled for age, gender, education, prior AI chat exposure, digital literacy, technology anxiety, and assistance mode.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-7"><title>Heterogeneity and Sensitivity Analyses</title><p>The condition-by-digital literacy interaction for verification intention was near zero (&#x03B2;=&#x2212;0.005, 95% CI &#x2013;0.220 to 0.210; <italic>P</italic>=.96), suggesting that the Safety UI bundle effect on verification intention was broadly similar across the observed literacy range. To evaluate whether retaining participants who required family or interviewer assistance materially changed the observed intervention effect, we repeated the analysis among participants who completed the survey independently (n=134). The pattern remained consistent, with mean verification intention 4.50 (SD 0.58) in the Baseline UI arm and 4.70 (SD 0.66) in the Safety UI bundle arm. In the covariate-adjusted model restricted to independent completers, assignment to the Safety UI bundle was associated with higher verification intention (adjusted &#x03B2;=0.218, 95% CI &#x2212;0.025 to 0.461; <italic>P</italic>=.08). The estimate was directionally consistent with the primary analysis, although the CI included zero.</p><p>Sensitivity analyses using complete-case data were identical to the main analytic sample because missingness did not exceed the prespecified threshold. Gender and education showed the most visible between-arm differences; because both were prespecified covariates in the primary adjusted models, these imbalances were already accounted for, and no separate post hoc adjustment was required.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>The safety UI bundle was associated with higher verification intention on a 7-point scale, with an adjusted difference of about 0.29 and a moderate standardized effect. Reliance intention did not increase and showed a small, nonsignificant decrease, indicating that the bundle shifted users toward more verification without encouraging greater direct reliance.</p></sec><sec id="s4-2"><title>Comparison With Current Literature</title><p>The observed increase in verification intention with the safety UI bundle aligns with a growing body of human-centered AI and automation trust research demonstrating that interface cues, such as uncertainty indicators, source displays, or explicit verification prompts, can meaningfully shift user behavior toward more critical engagement and reduce blind reliance on AI outputs [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref21">21</xref>]. Prior studies have shown that while explanations and transparency features often increase users&#x2019; trust in AI systems, they do not always translate into greater verification or scrutiny of outputs [<xref ref-type="bibr" rid="ref14">14</xref>-<xref ref-type="bibr" rid="ref16">16</xref>]. In contrast, direct verification nudges, such as prompts to check sources or warnings about potential errors, have been found to be more effective at encouraging users to verify information rather than simply increasing their confidence in the system [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>]. The effect size observed here is comparable to those reported for other UI interventions aimed at mitigating misinformation or promoting critical evaluation, such as warning labels and uncertainty communication strategies [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref22">22</xref>]. Notably, such interventions can shift intentions even when the underlying content quality remains unchanged, underscoring the value of interface-level design for safety [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>].</p><p>This pattern supports trust calibration frameworks emphasizing that safe adoption of AI is not about simply increasing or decreasing trust but about aligning user reliance with system uncertainty and capability [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref24">24</xref>-<xref ref-type="bibr" rid="ref26">26</xref>]. Overreliance, sometimes termed &#x201C;automation bias,&#x201D; is a well-documented risk when systems appear authoritative or present information without cues about limitations [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref28">28</xref>]. By explicitly framing uncertainty and encouraging cross-checking, the safety UI bundle likely counteracted authority cues that can lead to excessive deference to AI recommendations [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]. Similar effects have been observed in clinical decision support and navigation aids, where calibrated trust interventions reduced overreliance without undermining appropriate use [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref30">30</xref>]. The present findings extend this literature by demonstrating that such calibration can be achieved through simple UI modifications in a vignette setting with older adults, a group often considered at higher risk for overtrust due to lower digital literacy or unfamiliarity with automated systems [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref31">31</xref>].</p><p>The modest increase in behavioral verification proxies echoes prior research documenting gaps between stated intentions (self-report) and observable behaviors in digital environments [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>]. Behavioral measures in survey-based experiments often yield smaller effects and greater variance than self-reported scales, particularly when the behavior is low-cost (eg, clicking to expand a source) and optional [<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>]. Stronger behavioral endpoints, such as actual web searching, time spent reviewing sources, or correctness-based incentives, tend to produce more robust effects but are less feasible in vignette studies [<xref ref-type="bibr" rid="ref23">23</xref>,<xref ref-type="bibr" rid="ref32">32</xref>]. Nonetheless, prior work suggests that even modest, noisy behavioral uplifts can support the directionality of intervention effects, especially when triangulated with self-report and manipulation checks [<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>]. The findings highlight both the promise and limitations of using lightweight behavioral proxies in survey experiments and underscore the need for richer behavioral tracking in future research [<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>].</p><p>This result contrasts with some studies on warning labels and uncertainty displays that report reduced comprehension or increased confusion, particularly among older adults or users with lower literacy [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref35">35</xref>]. In other cases, disclaimers or uncertainty cues have distracted users or reduced perceived clarity of information [<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref35">35</xref>]. The present design may have avoided the pitfalls by employing large-font screenshots, a stable layout, and concise cues placed below the answer, features known to support comprehension and minimize cognitive load for older users [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref31">31</xref>]. While prior work sometimes finds that added UI elements increase cognitive load or reduce clarity, the nonworsening comprehension observed here suggests that conservative answer structure and scenario simplicity can help preserve understanding even when safety cues are introduced [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref34">34</xref>].</p><p>The observed reduction in perceived trustworthiness is consistent with research showing that explicit uncertainty communication can lower perceived competence or reliability of AI systems, even as it improves trust calibration and reduces overreliance [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>]. Citation displays alone often increase perceived credibility and may inadvertently encourage overtrust; however, bundling citations with an uncertainty nudge appears to temper this effect by signaling to users that outputs should be verified rather than accepted at face value [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref19">19</xref>]. This aligns with literature distinguishing &#x201C;trust&#x201D; from &#x201C;appropriate trust,&#x201D; arguing that a small reduction in perceived trustworthiness may be acceptable, or even desirable, if it prevents overreliance on potentially fallible systems [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]. Similar trade-offs have been reported in studies of warning labels, misinformation interventions, and human-automation warning systems where increased vigilance comes at the cost of slightly diminished confidence in system outputs [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref22">22</xref>].</p><p>The findings are encouraging given gerontechnology and accessibility research emphasizing that additional UI elements can burden older adults, especially on small screens or in chat interfaces [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref34">34</xref>]. Prior studies have found that added explanations or complex provenance panels can reduce usability scores among older users or those with limited digital literacy [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref16">16</xref>]. The cues used here were intentionally short, visually distinct, and placed in predictable locations, a design approach similar to minimalistic &#x201C;just-in-time&#x201D; warnings shown to preserve usability while enhancing safety in HCI research [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref36">36</xref>]. If previous work has argued that safety cues harm user experience, the null usability differences suggest that careful cue design can maintain user satisfaction even as verification orientation is improved [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref36">36</xref>].</p><p>High rates of cue recognition bolster confidence in the internal validity of the experimental manipulation. Prior UI cue experiments have sometimes endured manipulation failure, where participants do not notice disclaimers or misunderstand uncertainty indicators, undermining inference about intervention effects [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref19">19</xref>]. Salience and comprehensibility of cues are especially critical for older populations who may miss subtle interface changes due to sensory or cognitive constraints [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref31">31</xref>]. The rates of cue recognition observed here appear comparable to those reported for effective labels and warnings in digital interfaces; design choices such as clear visual separation and concise language likely contributed to this salience without increasing burden [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref36">36</xref>].</p><p>While some studies suggest low-literacy users benefit more from explicit guidance or verification prompts [<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref37">37</xref>], no meaningful moderation was observed here. Possible explanations include a restricted range of digital literacy within the sample, assistance reducing effective literacy gaps during completion, or the simplicity of the cue making it broadly effective across subgroups [<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref37">37</xref>]. In contrast to literature showing strong moderation by education or eHealth literacy on intervention effects [<xref ref-type="bibr" rid="ref37">37</xref>], the results suggest that well-designed safety UI bundles may be broadly applicable. Nevertheless, targeted tests in more diverse or lower-literacy samples remain warranted to confirm generalizability [<xref ref-type="bibr" rid="ref37">37</xref>].</p></sec><sec id="s4-3"><title>Strengths, Limitations, and Implications</title><p>This study used randomized assignment to estimate the effect of a practical safety UI bundle on verification orientation in a generative AI chat interface, while holding prompt and answer content constant. The design targets older Chinese adults, a population with high potential benefit from generative AI assistance but potential vulnerability to automation bias, low digital literacy, and difficulty evaluating AI-generated information.</p><p>Several limitations should be considered. Outcomes were based primarily on self-reported intentions within a vignette screenshot setting rather than observed real-world behavior in deployed systems. Static screenshots may understate the cognitive demands of real-time generative AI interaction, where users formulate follow-up questions, interpret changing responses, and decide whether to verify under time pressure. The behavioral proxy was minimal and may not capture substantive verification actions such as checking external sources or consulting professionals. Because assisted completion was permitted for accessibility, some participants with visual barriers may have received the source labels and the uncertainty/verification nudge through verbatim oral reading by a family member or interviewer. For these participants, the intervention may have functioned partly as a verbally mediated warning rather than a purely visual UI nudge. Although assistance mode was recorded, adjusted for, and examined in sensitivity analyses, the study cannot fully isolate visual salience from orally mediated cue exposure among assisted participants.</p><p>The study tested a bundle rather than individual safety cues, so it cannot determine whether source labels, uncertainty messaging, or verification nudges drove the observed effects. The generic source chips were useful for avoiding false provenance but may have reduced ecological realism by removing the cognitive burden of evaluating real citations. Because the verification nudge explicitly recommended checking authoritative sources and consulting professionals, verification-intention items may have been susceptible to demand characteristics or social desirability bias. The 2 scenarios were intentionally low risk and conservatively written, which may have produced high comprehension scores and limited the ability to detect comprehension trade-offs during more complex, ambiguous, or error-containing AI responses. The answers did not contain false or hallucinated information, so reduced reliance may partly reflect undertrust of accurate content rather than trust calibration in the strict sense. The cross-sectional design cannot assess habituation, banner blindness, or long-term use. Assistance during completion may have altered subjective cognitive load or comprehension, and on-site staff were not formally blinded after assignment, creating a potential source of interviewer bias. Residual confounding from baseline imbalances, including education and gender, may remain despite covariate adjustment. Finally, the sample size was selected for the continuous primary outcome and was likely underpowered for the binary expansion outcome. Additionally, because several primary or supportive outcomes were tested and no formal alpha adjustment was applied, the possibility of inflated Type I error should be considered.</p><p>The findings suggest that simple, implementable safety cues may shift older users toward more verification without degrading comprehension or perceived usability, although perceived trustworthiness may decrease modestly. Design and development teams can treat verification nudges and transparent framing of uncertainty as promising directions for further testing, while recognizing that effects may depend on wording, timing, interface prominence, user characteristics, and whether real citations or interactive system behavior are present.</p></sec><sec id="s4-4"><title>Future Work</title><p>Future work should test interactive generative AI systems, real verification behavior outside the survey environment, repeated exposure over time, and factorial designs that disentangle source labels, uncertainty language, and verification prompts. Studies should also compare generic labels with real citations to determine whether added ecological realism changes cognitive load, comprehension, and verification behavior.</p></sec><sec id="s4-5"><title>Conclusions</title><p>In this randomized static-vignette survey, a safety-oriented UI bundle in a generative AI chat interface was associated with higher verification intention among older Chinese adults, while reliance intention, comprehension, and usability/readability remained largely unchanged. The trust calibration index moved in a direction consistent with lower overreliance risk, but interpretation should remain cautious because the study did not include incorrect AI outputs and was not designed to isolate individual UI components.</p></sec></sec></body><back><ack><p>KimiChat was used to translate the initial Chinese draft into English. ChatGPT was used for linguistic proofreading of the final version prepared for submission, subsequent manuscript revisions, and author responses. All AI-assisted outputs were reviewed, verified, and edited by the authors, who remain fully responsible for the accuracy, integrity, and final content of the manuscript.</p></ack><notes><sec><title>Funding</title><p>This study received no external funding.</p></sec><sec><title>Data Availability</title><p>The deidentified data analyzed in the study may be provided by the corresponding author on reasonable request, subject to ethics approval requirements and protection of participant privacy. The analytic code may be shared on reasonable request where permitted by institutional policy.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: JY, JC, AR, ZG</p><p>Data curation: JY, HM</p><p>Formal analysis: JY</p><p>Investigation: JY, JC, HD</p><p>Methodology: JY, JC, AR, HD, ZG</p><p>Project administration: AR</p><p>Resources: JC, HD</p><p>Supervision: AR, ZG</p><p>Validation: HD, ZG</p><p>Writing - original draft: JY, AR</p><p>Writing - review &#x0026; editing: JY, JC, AR, HD, HM, ZG</p><p>All authors reviewed and approved the final manuscript</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">HCI</term><def><p>human-computer interaction</p></def></def-item><def-item><term id="abb2">OR</term><def><p>odds ratio</p></def></def-item><def-item><term id="abb3">UI</term><def><p>user interface</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hyun Baek</surname><given-names>T</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>M</given-names> </name></person-group><article-title>Is ChatGPT scary good? How user motivations affect creepiness and trust in generative artificial intelligence</article-title><source>Telemat Inform</source><year>2023</year><month>09</month><volume>83</volume><fpage>102030</fpage><pub-id pub-id-type="doi">10.1016/j.tele.2023.102030</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Weisz</surname><given-names>JD</given-names> </name><name name-style="western"><surname>He</surname><given-names>J</given-names> </name><name name-style="western"><surname>Muller</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hoefer</surname><given-names>G</given-names> </name><name name-style="western"><surname>Miles</surname><given-names>R</given-names> </name><name name-style="western"><surname>Geyer</surname><given-names>W</given-names> </name></person-group><article-title>Design principles for generative AI applications</article-title><year>2024</year><month>05</month><day>11</day><access-date>2026-08-02</access-date><conf-name>Proceedings of the 2024 CHI Conference on Human Factors in Computing Systems</conf-name><conf-date>May 11, 2024</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/proceedings/10.1145/3613904">https://dl.acm.org/doi/proceedings/10.1145/3613904</ext-link></comment><pub-id pub-id-type="doi">10.1145/3613904.3642466</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Mayerhofer</surname><given-names>K</given-names> </name><name name-style="western"><surname>Capra</surname><given-names>R</given-names> </name><name name-style="western"><surname>Elsweiler</surname><given-names>D</given-names> </name></person-group><article-title>Blending queries and conversations: understanding trust, verification, and system choice in search and chat interactions</article-title><year>2025</year><month>03</month><day>24</day><access-date>2026-08-02</access-date><conf-name>CHIIR &#x2019;25</conf-name><conf-date>Mar 24, 2025</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/proceedings/10.1145/3698204">https://dl.acm.org/doi/proceedings/10.1145/3698204</ext-link></comment><pub-id pub-id-type="doi">10.1145/3698204.3716454</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Afroogh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Akbari</surname><given-names>A</given-names> </name><name name-style="western"><surname>Malone</surname><given-names>E</given-names> </name><name name-style="western"><surname>Kargar</surname><given-names>M</given-names> </name><name name-style="western"><surname>Alambeigi</surname><given-names>H</given-names> </name></person-group><article-title>Trust in AI: progress, challenges, and future directions</article-title><source>Humanit Soc Sci Commun</source><year>2024</year><volume>11</volume><issue>1</issue><pub-id pub-id-type="doi">10.1057/s41599-024-04044-8</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bach</surname><given-names>TA</given-names> </name><name name-style="western"><surname>Khan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hallock</surname><given-names>H</given-names> </name><name name-style="western"><surname>Beltr&#x00E3;o</surname><given-names>G</given-names> </name><name name-style="western"><surname>Sousa</surname><given-names>S</given-names> </name></person-group><article-title>A systematic literature review of user trust in AI-enabled systems: an HCI perspective</article-title><source>Int J Hum Comput Interact</source><year>2024</year><month>03</month><day>3</day><volume>40</volume><issue>5</issue><fpage>1251</fpage><lpage>1266</lpage><pub-id pub-id-type="doi">10.1080/10447318.2022.2138826</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gulati</surname><given-names>S</given-names> </name><name name-style="western"><surname>McDonagh</surname><given-names>J</given-names> </name><name name-style="western"><surname>Sousa</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lamas</surname><given-names>D</given-names> </name></person-group><article-title>Trust models and theories in human&#x2013;computer interaction: a systematic literature review</article-title><source>Comput Hum Behav Rep</source><year>2024</year><month>12</month><volume>16</volume><fpage>100495</fpage><pub-id pub-id-type="doi">10.1016/j.chbr.2024.100495</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Ko</surname><given-names>EG</given-names> </name><name name-style="western"><surname>Nanayakkara</surname><given-names>S</given-names> </name><name name-style="western"><surname>Huff</surname><given-names>EW</given-names>  <suffix>Jr</suffix></name></person-group><article-title>&#x201C;We need to avail ourselves of [genai] to enhance knowledge distribution&#x201D;: empowering older adults through genai literacy</article-title><year>2025</year><month>04</month><day>26</day><access-date>2026-08-02</access-date><conf-name>Proceedings of the Extended Abstracts of the CHI Conference on Human Factors in Computing Systems</conf-name><conf-date>Apr 25 to May 1, 2025</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/proceedings/10.1145/3706599">https://dl.acm.org/doi/proceedings/10.1145/3706599</ext-link></comment><pub-id pub-id-type="doi">10.1145/3706599.3720032</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Al-Somali</surname><given-names>SA</given-names> </name></person-group><article-title>Integrating artificial intelligence (AI) in healthcare: advancing older adults&#x2019; health management in Saudi Arabia through AI-powered chatbots</article-title><source>PeerJ Comput Sci</source><year>2025</year><volume>11</volume><fpage>e2773</fpage><pub-id pub-id-type="doi">10.7717/peerj-cs.2773</pub-id><pub-id pub-id-type="medline">40567759</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ismatullaev</surname><given-names>UVU</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>SH</given-names> </name></person-group><article-title>Review of the factors affecting acceptance of AI-infused systems</article-title><source>Hum Factors</source><year>2024</year><month>01</month><volume>66</volume><issue>1</issue><fpage>126</fpage><lpage>144</lpage><pub-id pub-id-type="doi">10.1177/00187208211064707</pub-id><pub-id pub-id-type="medline">35344676</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Fang</surname><given-names>G</given-names> </name></person-group><article-title>Factors influencing the acceptance of medical AI chat assistants among healthcare professionals and patients: a survey-based study in China</article-title><source>Front Public Health</source><year>2025</year><volume>13</volume><pub-id pub-id-type="doi">10.3389/fpubh.2025.1637270</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>SD</given-names> </name></person-group><article-title>Application and challenges of the technology acceptance model in elderly healthcare: insights from ChatGPT</article-title><source>Technologies (Basel)</source><year>2024</year><volume>12</volume><issue>5</issue><fpage>68</fpage><pub-id pub-id-type="doi">10.3390/technologies12050068</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schroeder</surname><given-names>T</given-names> </name><name name-style="western"><surname>Dodds</surname><given-names>L</given-names> </name><name name-style="western"><surname>Georgiou</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gewald</surname><given-names>H</given-names> </name><name name-style="western"><surname>Siette</surname><given-names>J</given-names> </name></person-group><article-title>Older adults and new technology: mapping review of the factors associated with older adults&#x2019; intention to adopt digital technologies</article-title><source>JMIR Aging</source><year>2023</year><month>05</month><day>16</day><volume>6</volume><fpage>e44564</fpage><pub-id pub-id-type="doi">10.2196/44564</pub-id><pub-id pub-id-type="medline">37191976</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Ahmadianmanzary</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ouhbi</surname><given-names>S</given-names> </name></person-group><article-title>Exploring the influence of user interface on user trust in generative AI</article-title><access-date>2026-08-02</access-date><conf-name>20th International Conference on Evaluation of Novel Approaches to Software Engineering</conf-name><conf-date>Apr 4-6, 2025</conf-date><conf-loc>Porto, Portugal</conf-loc><fpage>708</fpage><lpage>714</lpage><comment><ext-link ext-link-type="uri" xlink:href="http://www.scitepress.org/DigitalLibrary/ProceedingLink.aspx?ID=1893">http://www.scitepress.org/DigitalLibrary/ProceedingLink.aspx?ID=1893</ext-link></comment><pub-id pub-id-type="doi">10.5220/0013432700003928</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Naiseh</surname><given-names>M</given-names> </name><name name-style="western"><surname>Al-Thani</surname><given-names>D</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>N</given-names> </name><name name-style="western"><surname>Ali</surname><given-names>R</given-names> </name></person-group><article-title>How the different explanation classes impact trust calibration: the case of clinical decision support systems</article-title><source>Int J Hum Comput Stud</source><year>2023</year><month>01</month><volume>169</volume><fpage>102941</fpage><pub-id pub-id-type="doi">10.1016/j.ijhcs.2022.102941</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Leichtmann</surname><given-names>B</given-names> </name><name name-style="western"><surname>Humer</surname><given-names>C</given-names> </name><name name-style="western"><surname>Hinterreiter</surname><given-names>A</given-names> </name><name name-style="western"><surname>Streit</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mara</surname><given-names>M</given-names> </name></person-group><article-title>Effects of explainable artificial intelligence on trust and human behavior in a high-risk decision task</article-title><source>Comput Human Behav</source><year>2023</year><month>02</month><volume>139</volume><fpage>107539</fpage><pub-id pub-id-type="doi">10.1016/j.chb.2022.107539</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Naiseh</surname><given-names>M</given-names> </name><name name-style="western"><surname>Al-Thani</surname><given-names>D</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>N</given-names> </name><name name-style="western"><surname>Ali</surname><given-names>R</given-names> </name></person-group><article-title>Explainable recommendation: when design meets trust calibration</article-title><source>World Wide Web</source><year>2021</year><volume>24</volume><issue>5</issue><fpage>1857</fpage><lpage>1884</lpage><pub-id pub-id-type="doi">10.1007/s11280-021-00916-0</pub-id><pub-id pub-id-type="medline">34366701</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zerilli</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bhatt</surname><given-names>U</given-names> </name><name name-style="western"><surname>Weller</surname><given-names>A</given-names> </name></person-group><article-title>How transparency modulates trust in artificial intelligence</article-title><source>Patterns (N Y)</source><year>2022</year><month>04</month><day>8</day><volume>3</volume><issue>4</issue><fpage>100455</fpage><pub-id pub-id-type="doi">10.1016/j.patter.2022.100455</pub-id><pub-id pub-id-type="medline">35465233</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bu&#x00E7;inca</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Malaya</surname><given-names>MB</given-names> </name><name name-style="western"><surname>Gajos</surname><given-names>KZ</given-names> </name></person-group><article-title>To trust or to think</article-title><source>Proc ACM Hum-Comput Interact</source><year>2021</year><month>04</month><day>13</day><volume>5</volume><issue>CSCW1</issue><fpage>1</fpage><lpage>21</lpage><pub-id pub-id-type="doi">10.1145/3449287</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Liao</surname><given-names>QV</given-names> </name><name name-style="western"><surname>Sundar</surname><given-names>SS</given-names> </name></person-group><article-title>Designing for responsible trust in AI systems: a communication perspective</article-title><year>2022</year><month>06</month><day>21</day><access-date>2026-08-02</access-date><conf-name>FAccT &#x2019;22</conf-name><conf-date>Jun 21, 2022</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/proceedings/10.1145/3531146">https://dl.acm.org/doi/proceedings/10.1145/3531146</ext-link></comment><pub-id pub-id-type="doi">10.1145/3531146.3533182</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>T</given-names> </name></person-group><article-title>Understanding older adults&#x2019; acceptance of chatbots in healthcare delivery: an extended UTAUT model</article-title><source>Front Public Health</source><year>2024</year><volume>12</volume><fpage>1435329</fpage><pub-id pub-id-type="doi">10.3389/fpubh.2024.1435329</pub-id><pub-id pub-id-type="medline">39628811</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Choudhuri</surname><given-names>R</given-names> </name><name name-style="western"><surname>Trinkenreich</surname><given-names>B</given-names> </name><name name-style="western"><surname>Pandita</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kalliamvakou</surname><given-names>E</given-names> </name><name name-style="western"><surname>Steinmacher</surname><given-names>I</given-names> </name><name name-style="western"><surname>Gerosa</surname><given-names>M</given-names> </name><etal/></person-group><article-title>What needs attention? prioritizing drivers of developers&#x2019; trust and adoption of generative AI</article-title><source>ArXiv</source><comment>Preprint posted online on  Nov 14, 2025</comment><pub-id pub-id-type="doi">10.48550/arxiv.2505.17418</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>JY</given-names> </name><name name-style="western"><surname>Lester</surname><given-names>C</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>XJ</given-names> </name></person-group><article-title>Beyond binary decisions: evaluating the effects of AI error type on trust and performance in AI-assisted tasks</article-title><source>Hum Factors</source><year>2025</year><month>10</month><volume>67</volume><issue>10</issue><fpage>1062</fpage><lpage>1083</lpage><pub-id pub-id-type="doi">10.1177/00187208251326795</pub-id><pub-id pub-id-type="medline">40104968</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bezzaoui</surname><given-names>I</given-names> </name><name name-style="western"><surname>Stein</surname><given-names>C</given-names> </name><name name-style="western"><surname>Weinhardt</surname><given-names>C</given-names> </name><name name-style="western"><surname>Fegert</surname><given-names>J</given-names> </name></person-group><article-title>Explainable AI for online disinformation detection: Insights from a design science research project</article-title><source>Electron Markets</source><year>2025</year><month>12</month><volume>35</volume><issue>1</issue><pub-id pub-id-type="doi">10.1007/s12525-025-00799-3</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Klingbeil</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gr&#x00FC;tzner</surname><given-names>C</given-names> </name><collab>AI</collab></person-group><article-title>Trust and reliance on AI &#x2014; an experimental study on the extent and costs of overreliance on AI</article-title><source>Comput Human Behav</source><year>2024</year><month>11</month><volume>160</volume><fpage>108352</fpage><pub-id pub-id-type="doi">10.1016/j.chb.2024.108352</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>B</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Luan</surname><given-names>S</given-names> </name></person-group><article-title>Developing trustworthy artificial intelligence: insights from research on interpersonal, human-automation, and human-AI trust</article-title><source>Front Psychol</source><year>2024</year><volume>15</volume><pub-id pub-id-type="doi">10.3389/fpsyg.2024.1382693</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Middleton</surname><given-names>SE</given-names> </name><name name-style="western"><surname>Letouz&#x00E9;</surname><given-names>E</given-names> </name><name name-style="western"><surname>Hossaini</surname><given-names>A</given-names> </name><name name-style="western"><surname>Chapman</surname><given-names>A</given-names> </name></person-group><article-title>Trust, regulation, and human-in-the-loop AI</article-title><source>Commun ACM</source><year>2022</year><month>04</month><volume>65</volume><issue>4</issue><fpage>64</fpage><lpage>68</lpage><pub-id pub-id-type="doi">10.1145/3511597</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhai</surname><given-names>C</given-names> </name><name name-style="western"><surname>Wibowo</surname><given-names>S</given-names> </name><name name-style="western"><surname>Li</surname><given-names>LD</given-names> </name></person-group><article-title>The effects of over-reliance on AI dialogue systems on students&#x2019; cognitive abilities: a systematic review</article-title><source>Smart Learn Environ</source><year>2024</year><volume>11</volume><issue>1</issue><pub-id pub-id-type="doi">10.1186/s40561-024-00316-7</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>von Eschenbach</surname><given-names>WJ</given-names> </name></person-group><article-title>Transparency and the black box problem: why we do not trust AI</article-title><source>Philos Technol</source><year>2021</year><month>12</month><volume>34</volume><issue>4</issue><fpage>1607</fpage><lpage>1622</lpage><pub-id pub-id-type="doi">10.1007/s13347-021-00477-0</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Gabriel</surname><given-names>I</given-names> </name><name name-style="western"><surname>Manzini</surname><given-names>A</given-names> </name><name name-style="western"><surname>Keeling</surname><given-names>G</given-names> </name><name name-style="western"><surname>Hendricks</surname><given-names>L</given-names> </name><name name-style="western"><surname>Rieser</surname><given-names>V</given-names> </name><name name-style="western"><surname>Iqbal</surname><given-names>H</given-names> </name><etal/></person-group><article-title>The ethics of advanced AI assistants</article-title><source>ArXiv</source><comment>Preprint posted online on  Apr 28, 2024</comment><pub-id pub-id-type="doi">10.48550/arxiv.2404.16244</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Labkoff</surname><given-names>S</given-names> </name><name name-style="western"><surname>Oladimeji</surname><given-names>B</given-names> </name><name name-style="western"><surname>Kannry</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Toward a responsible future: recommendations for AI-enabled clinical decision support</article-title><source>J Am Med Inform Assoc</source><year>2024</year><month>11</month><day>1</day><volume>31</volume><issue>11</issue><fpage>2730</fpage><lpage>2739</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocae209</pub-id><pub-id pub-id-type="medline">39325508</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Hou</surname><given-names>G</given-names> </name></person-group><article-title>The impact of usage experience and input modality on trust experience and cognitive load in older adults</article-title><source>Front Comput Sci</source><year>2025</year><volume>7</volume><pub-id pub-id-type="doi">10.3389/fcomp.2025.1659594</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Choudhury</surname><given-names>A</given-names> </name><name name-style="western"><surname>Shamszare</surname><given-names>H</given-names> </name></person-group><article-title>Investigating the impact of user trust on the adoption and use of ChatGPT: survey analysis</article-title><source>J Med Internet Res</source><year>2023</year><month>06</month><day>14</day><volume>25</volume><fpage>e47184</fpage><pub-id pub-id-type="doi">10.2196/47184</pub-id><pub-id pub-id-type="medline">37314848</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Leschanowsky</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rech</surname><given-names>S</given-names> </name><name name-style="western"><surname>Popp</surname><given-names>B</given-names> </name><name name-style="western"><surname>B&#x00E4;ckstr&#x00F6;m</surname><given-names>T</given-names> </name></person-group><article-title>Evaluating privacy, security, and trust perceptions in conversational AI: a systematic review</article-title><source>Comput Human Behav</source><year>2024</year><month>10</month><volume>159</volume><fpage>108344</fpage><pub-id pub-id-type="doi">10.1016/j.chb.2024.108344</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sadeghi</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Alizadehsani</surname><given-names>R</given-names> </name><name name-style="western"><surname>Cifci</surname><given-names>MA</given-names> </name><etal/></person-group><article-title>A review of explainable artificial intelligence in healthcare</article-title><source>Comput Electr Eng</source><year>2024</year><month>08</month><volume>118</volume><fpage>109370</fpage><pub-id pub-id-type="doi">10.1016/j.compeleceng.2024.109370</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Virvou</surname><given-names>M</given-names> </name><name name-style="western"><surname>Tsihrintzis</surname><given-names>GA</given-names> </name><name name-style="western"><surname>Tsichrintzi</surname><given-names>EA</given-names> </name></person-group><article-title>VIRTSI: a novel trust dynamics model enhancing artificial intelligence collaboration with human users &#x2013; insights from a ChatGPT evaluation study</article-title><source>Inf Sci (Ny)</source><year>2024</year><month>07</month><volume>675</volume><fpage>120759</fpage><pub-id pub-id-type="doi">10.1016/j.ins.2024.120759</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Al Ansari</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Al Ahmed</surname><given-names>Y</given-names> </name><name name-style="western"><surname>El Bahnaswi</surname><given-names>HH</given-names> </name></person-group><article-title>Balancing usability and protection in AI and data security: a human-centric approach</article-title><conf-name>2024 11th International Conference on Software Defined Systems (SDS)</conf-name><conf-date>Dec 9, 2024</conf-date><conf-loc>Gran Canaria, Spain</conf-loc><fpage>80</fpage><lpage>88</lpage><pub-id pub-id-type="doi">10.1109/SDS64317.2024.10883898</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Acosta-Enriquez</surname><given-names>BG</given-names> </name><name name-style="western"><surname>Arbul&#x00FA; Ballesteros</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Huaman&#x00ED; Jordan</surname><given-names>O</given-names> </name><name name-style="western"><surname>L&#x00F3;pez Roca</surname><given-names>C</given-names> </name><name name-style="western"><surname>Saavedra Tirado</surname><given-names>K</given-names> </name></person-group><article-title>Analysis of college students&#x2019; attitudes toward the use of ChatGPT in their academic activities: effect of intent to use, verification of information and responsible use</article-title><source>BMC Psychol</source><year>2024</year><month>05</month><day>8</day><volume>12</volume><issue>1</issue><fpage>255</fpage><pub-id pub-id-type="doi">10.1186/s40359-024-01764-z</pub-id><pub-id pub-id-type="medline">38720382</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Stimulus screenshots of the baseline and safety user interface (UI) conditions.</p><media xlink:href="jmir_v28i1e94140_app1.pdf" xlink:title="PDF File, 912 KB"/></supplementary-material></app-group></back></article>