<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">J Med Internet Res</journal-id><journal-id journal-id-type="publisher-id">jmir</journal-id><journal-id journal-id-type="index">1</journal-id><journal-title>Journal of Medical Internet Research</journal-title><abbrev-journal-title>J Med Internet Res</abbrev-journal-title><issn pub-type="epub">1438-8871</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v28i1e91516</article-id><article-id pub-id-type="doi">10.2196/91516</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Health-Related Rumor Debunking on Sina Weibo in China (2009-2024): 15-Year Retrospective Infodemiology Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Fang</surname><given-names>Yuan</given-names></name><degrees>BMgmt</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>He</surname><given-names>Chengwu</given-names></name><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hu</surname><given-names>Yejinxuan</given-names></name><degrees>BMgmt</degrees><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Tian</surname><given-names>Xianyun</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1"/></contrib></contrib-group><aff id="aff1"><institution>College of Management Science, Chengdu University of Technology</institution><addr-line>No. 1, East Third Road, Erxianqiao, Chenghua District</addr-line><addr-line>Chengdu</addr-line><addr-line>Sichuan</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Mavragani</surname><given-names>Amaryllis</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Balogun</surname><given-names>Babatunde</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Yiannakoulias</surname><given-names>Nikolaos</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Guo</surname><given-names>Rui</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Xianyun Tian, PhD, College of Management Science, Chengdu University of Technology, No. 1, East Third Road, Erxianqiao, Chenghua District, Chengdu, Sichuan, 610059, China, 86 17794550089; <email>xianyuntian@cdut.edu.cn</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>10</day><month>9</month><year>2026</year></pub-date><volume>28</volume><elocation-id>e91516</elocation-id><history><date date-type="received"><day>29</day><month>01</month><year>2026</year></date><date date-type="rev-recd"><day>04</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>15</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Yuan Fang, Chengwu He, Yejinxuan Hu, Xianyun Tian. Originally published in the Journal of Medical Internet Research (<ext-link ext-link-type="uri" xlink:href="https://www.jmir.org">https://www.jmir.org</ext-link>), 10.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Journal of Medical Internet Research (ISSN 1438-8871), is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.jmir.org/">https://www.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://www.jmir.org/2026/1/e91516"/><abstract><sec><title>Background</title><p>Understanding how health-related rumor debunking evolves and spreads on social media is critical for public health communication and policy. Existing research, however, has been largely crisis-centered&#x2014;dominated by studies of specific events, such as the COVID-19 pandemic&#x2014;and offers limited insight into the longer-term patterns of thematic evolution, demographic targeting, and engagement dynamics of official debunking practices.</p></sec><sec><title>Objective</title><p>This study aimed to provide an integrated understanding of official health rumor debunking in China over a 15-year period by delineating its thematic evolution, demographic disparities, and features associated with engagement and information diffusion.</p></sec><sec sec-type="methods"><title>Methods</title><p>We collected rumor debunking posts published on Sina Weibo between 2009 and 2024. A 2-stage classification pipeline using Sentence Transformer Fine-Tuning was constructed to identify health-related debunking posts. We applied BERTopic (BERT: Bidirectional Encoder Representations from Transformers) to map the thematic landscape and used a large language model to extract demographic mentions (gender, age, and social roles), and health content referenced in the posts. Finally, we used Extreme Gradient Boosting regression models with Shapley Additive Explanations to quantify the relative contributions of predictors to engagement and information diffusion, incorporating a multidimensional feature space spanning user metadata, content characteristics, and temporal and contextual attributes.</p></sec><sec sec-type="results"><title>Results</title><p>Across 377,520 rumor debunking posts from 24,742 blue-verified accounts, 93,799 posts (24.8%) were health-related. Health debunking volume surged during the COVID-19 pandemic and remained elevated above prepandemic levels through 2024, although its relative share declined markedly in 2024, indicating a shift away from pandemic-centered topics, even as overall debunking activity continued to rise. Prevalent themes included vaccines, debunking reports, personal care, and sleep hygiene. Temporal trajectories fell into 3 patterns: event-driven spikes, sustained growth, and recurrent fluctuations. Demographic mentions were uneven: women were referenced more often than men, youth were the most frequently mentioned age group, and older adults were the least mentioned. Engagement was predicted primarily by source-level reach: follower count was the strongest predictor of reposts, comments, and likes alike, showing a nonlinear, threshold-like pattern in each case, whereas the secondary predictors differed across the 3 interaction types.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>This 15-year longitudinal analysis shows that health-related debunking on Sina Weibo was strongly shaped by major public health crises, surging during the COVID-19 pandemic and remaining elevated, while broadening toward routine, lifestyle-related concerns. Across all interaction types, engagement was shaped more by source reach than by the content of corrections. Although most posts addressed the general public, those with explicit demographic references revealed uneven representation across gender and age groups, with distinct health concerns linked to each. Taken together, these findings reveal a structural asymmetry in the debunking ecosystem: content is diversifying while distribution remains governed by source-level reach, suggesting that platform-level mechanisms that help credible but less-followed sources circulate corrections may be valuable for amplifying reliable health information.</p></sec></abstract><kwd-group><kwd>infodemiology</kwd><kwd>health misinformation</kwd><kwd>rumor debunking</kwd><kwd>social media</kwd><kwd>health communication</kwd><kwd>public health surveillance</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>The rapid expansion of the internet has lowered barriers to information dissemination, enabling content to circulate through online networks at unprecedented speed [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. In such environments, rumors, unverified claims that spread widely in the absence of conclusive evidence, can proliferate, distort public opinion, amplify anxiety, and, in some cases, lead to serious real-world consequences [<xref ref-type="bibr" rid="ref3">3</xref>-<xref ref-type="bibr" rid="ref7">7</xref>].</p><p>Among the many forms of online misinformation, health-related rumors are particularly concerning because they involve issues closely tied to individual safety and public well-being [<xref ref-type="bibr" rid="ref8">8</xref>]. At the same time, the specialized nature of medical knowledge often makes it difficult for nonexperts to assess information credibility, rendering many users vulnerable to misleading claims [<xref ref-type="bibr" rid="ref9">9</xref>-<xref ref-type="bibr" rid="ref13">13</xref>]. When such misleading claims take root among the public, they can disrupt public health decision-making, erode institutional trust, and produce severe real-world consequences [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. During the COVID-19 pandemic, for example, false claims that consuming methanol or alcohol-based disinfectants could eliminate the virus resulted in hundreds of preventable deaths and thousands of hospitalizations worldwide [<xref ref-type="bibr" rid="ref16">16</xref>].</p><p>To mitigate these risks, governments, public health agencies, social media platforms, and news organizations have increasingly adopted rumor debunking strategies to curb the spread of misleading health information and restore public trust [<xref ref-type="bibr" rid="ref17">17</xref>-<xref ref-type="bibr" rid="ref19">19</xref>]. On social media platforms, official debunking posts play an increasingly important role in countering misinformation by providing corrective information from authoritative sources and helping to mitigate the adverse effects of false claims [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>]. Rather than merely reacting to individual falsehoods, these debunking messages function as systematic institutional interventions embedded within broader information governance structures. As such corrective communication becomes increasingly central to public health, understanding how health-related debunking operates in real-world social media environments has emerged as an urgent priority for both researchers and practitioners.</p><p>Existing research on health misinformation and rumor debunking has advanced along several complementary lines. Prior studies have examined the thematic characteristics of health misinformation, identifying recurring issue domains such as public health crises, diseases, and diet- and nutrition-related claims [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. Other work has investigated who is susceptible to health misinformation, highlighting the roles of demographic and psychological factors in shaping vulnerability to misleading information [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref24">24</xref>]. Researchers have also explored how misinformation spreads across social media platforms, documenting diffusion dynamics and network mechanisms [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref25">25</xref>]. In parallel, substantial attention has been devoted to misinformation governance, including automated identification and monitoring systems [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref26">26</xref>], evaluations of corrective and debunking strategies [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>], and the mechanisms and temporal durability of prebunking (inoculation) interventions [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref27">27</xref>]. Finally, researchers have examined audience responses to corrections, identifying factors that influence the acceptance of or resistance to debunking messages [<xref ref-type="bibr" rid="ref18">18</xref>,<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]. Despite this growing body of literature, several important gaps remain.</p><p>First, existing studies on health-related debunking have largely focused on specific public health crises&#x2014;most notably COVID-19&#x2014;thereby providing episodic snapshots rather than a longitudinal understanding of evolving discourse [<xref ref-type="bibr" rid="ref15">15</xref>]. For example, Yang et al [<xref ref-type="bibr" rid="ref30">30</xref>] examined rumor characteristics during the COVID-19 pandemic, whereas Sicilia et al [<xref ref-type="bibr" rid="ref31">31</xref>] focused on the Zika virus. Although such crisis-centered approaches provide valuable insights into debunking dynamics under exceptional conditions, they are typically confined to relatively short time windows. Consequently, existing studies still lack a macro-level understanding of how health debunking themes are structured and how they evolve over time on social media platforms. Developing such a longitudinal perspective is important for informing the strategic allocation of debunking resources and supporting long-term public health governance planning [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>].</p><p>Second, prior research has carefully examined who is susceptible to health misinformation. Li and Shu [<xref ref-type="bibr" rid="ref34">34</xref>], for instance, identified demographic attributes such as gender and age as key determinants of vulnerability. Beyond demographics, cognitive traits have also been given substantial attention: Chua and Banerjee [<xref ref-type="bibr" rid="ref35">35</xref>] highlighted the vulnerability of &#x201C;na&#x00EF;ve believers&#x201D; who view knowledge as static, while Pennycook and Rand [<xref ref-type="bibr" rid="ref36">36</xref>] suggested that susceptibility is often associated with cognitive miserliness, particularly among individuals who rely on intuition rather than analytic reasoning. However, much of this evidence is derived from self-reported surveys or small-scale experiments. As a result, existing findings may lack ecological validity and may not fully capture which demographic groups are explicitly referenced in real-world debunking narratives or what kinds of health concerns are associated with them within platform-based information ecosystems [<xref ref-type="bibr" rid="ref37">37</xref>]. In practice, understanding which groups are explicitly mentioned as targets or affected audiences and what health risks are associated with these groups is crucial for precision debunking, improving intervention effectiveness, and optimizing resource allocation.</p><p>Third, existing research on the diffusion of health debunking information has primarily focused on differences across debunking content types, determinants of user engagement, and variations in debunking strategies. For example, Yang et al [<xref ref-type="bibr" rid="ref30">30</xref>] examined the effects of rumor categories (dread vs wish) and rhetorical strategies, while Chua and Banerjee [<xref ref-type="bibr" rid="ref35">35</xref>] investigated how different formats (text vs image) influence sharing intentions depending on user epistemic beliefs. Methodologically, however, this line of work predominantly relies on linear statistical models, such as regression analysis. While useful for estimating average effects, these approaches are often ill-suited for the skewed and nonlinear nature of social media diffusion data. As Sicilia et al [<xref ref-type="bibr" rid="ref31">31</xref>] noted, online information diffusion is shaped by complex network structures and long-tail distributions of influence (eg, follower counts), conditions that frequently violate the assumptions of traditional regression models. Consequently, how multidimensional factors&#x2014;spanning source, content, and contextual attributes&#x2014;jointly predict different modes of public engagement remains insufficiently understood. Uncovering these nonlinear mechanisms is crucial for platforms to optimize resource allocation and enhance the visibility of corrective communication.</p><p>Against this backdrop, this study conducted a systematic analysis of health-related debunking posts published by verified accounts on Sina Weibo (Beijing Weimeng Chuangke Network Technology Co., Ltd.), covering the period from August 2009 to December 2024. Notably, although these 3 research gaps have each attracted independent scholarly attention, existing studies have rarely examined them in conjunction. Yet the 3 dimensions correspond to successive stages of the same corrective communication process, from what is produced, to whom it addresses, to how far it travels, and the overall functioning of the debunking ecosystem may depend on their interaction rather than on any single dimension alone. Through this large-scale longitudinal investigation, we therefore aimed to advance an integrated, macro-level understanding of official health debunking practices in China by addressing the following research questions (RQs):</p><list list-type="bullet"><list-item><p>RQ1: What characterizes the thematic landscape of official health debunking in China, and how have these themes evolved over the past 15 years?</p></list-item><list-item><p>RQ2: Who are the targeted and affected groups mentioned in official health rumor debunking posts, and how do the associated health risks and concerns differ across these groups?</p></list-item><list-item><p>RQ3: What are the primary predictors of public engagement with official health debunking information, and how do their marginal contributions differ across interaction modes such as reposts, comments, and likes?</p></list-item></list><p>By addressing these questions, this study not only offers a comprehensive account of China&#x2019;s official health debunking efforts within a rapidly evolving social media environment but also provides actionable insights for public health authorities and platform administrators seeking to strengthen precise and efficient governance of health misinformation.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>This study adopted a retrospective infodemiology design to examine health-related rumor debunking posts on Sina Weibo over a period of 15 years (2009&#x2010;2024). Our analytical framework integrates natural language processing, large language models (LLMs), and interpretable machine learning to address three core dimensions: (1) the thematic evolution of health-related debunking, (2) the demographic profiles of the groups targeted or affected in debunking narratives, and (3) the key predictors associated with information diffusion and user engagement. The overall research workflow is summarized in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Schematic of the research framework. BERT: Bidirectional Encoder Representations from Transformers; LLM: large language modeling; SHAP: Shapley Additive Explanations; SetFit: Sentence Transformer Fine-Tuning.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e91516_fig01.png"/></fig></sec><sec id="s2-2"><title>Data Acquisition and Prefiltering</title><p>Sina Weibo is one of China&#x2019;s most influential social media platforms, reporting 587 million monthly active users [<xref ref-type="bibr" rid="ref38">38</xref>]. Its large-scale and diverse user base make it a valuable source for examining public discourse related to health rumor debunking.</p><p>Using the Chinese keyword &#x201C;rumor debunking&#x201D; (&#x8F9F;&#x8C23;), we retrieved posts published between August 1, 2009, and December 31, 2024. The initial corpus comprised 4,522,941 posts authored by 24,801 users.</p><p>To improve data quality, we applied a multistep filtering pipeline. First, based on Weibo&#x2019;s verification labels, we retained 493,956 posts published by &#x201C;Blue V&#x201D; (blue-verified) accounts, which are officially certified government agencies, media outlets, and organizational users. This restriction was motivated by prior evidence that official entities tend to provide more authoritative and effective rumor corrections [<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref40">40</xref>]. Second, we cleaned the data by removing 80,416 posts that either did not contain the keyword or fell outside the specified time window, yielding 413,540 candidate posts. Third, we applied classifiers, as detailed in the &#x201C;Classification Strategy and Model Selection&#x201D; section, to identify genuine debunking posts and further categorize them into health-related versus other topics.</p><p>Subsequently, we retrieved user-level metadata using the user IDs associated with the identified genuine debunking posts. For each post, we compiled a comprehensive feature set. Post-level attributes included the text content, publication time, and engagement metrics (likes, comments, and reposts), whereas user-level metadata (when available) included user ID, nickname, profile description, verification type, follower and following counts, and membership level. Account metadata for 948 users (1723 posts) were unavailable due to account anomalies (eg, self-deactivation or platform restrictions), but the corresponding posts were retained to preserve coverage for textual and thematic analyses. Ultimately, the final dataset contained 377,520 posts from 24,742 blue-verified accounts, including 93,799 health-related rumor debunking posts and 283,721 non&#x2013;health-related rumor debunking posts. For the analysis of publisher verification types, posts with missing or invalid verification-type information were excluded, resulting in a subset of 282,039 posts. Publisher verification-type distributions between health-related and non-health-related posts were compared using a Pearson chi-square test, with Cram&#x00E9;r <italic>V</italic> used to quantify the effect size.</p></sec><sec id="s2-3"><title>Classification Strategy and Model Selection</title><sec id="s2-3-1"><title>Overview</title><p>Following standard practices in text classification [<xref ref-type="bibr" rid="ref41">41</xref>], we adopted a 2-stage pipeline to improve task tractability and classification accuracy. Specifically, we first identified genuine rumor debunking posts from the full corpus and then classified the identified debunking posts as health related versus non&#x2013;health related. This staged design reduces task complexity and enables models to learn more discriminative features, thereby improving overall performance [<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref43">43</xref>]. To ensure the reliability of this pipeline, we relied on 2 core components: rigorous human annotation to establish high-quality gold-standard labels and systematic model evaluation to select the most robust classifier.</p></sec><sec id="s2-3-2"><title>Annotation Procedure and Label Definitions</title><p>We developed the annotation guidelines through an iterative process. The first author drafted an initial guideline after reviewing 500 randomly sampled posts. Next, the first and second authors completed multiple rounds of independent annotation, each round using a newly sampled set of 500 posts generated with different random seeds. The guideline was refined after each round until disagreements were largely eliminated and the criteria stabilized.</p><p>A post was labeled as debunking if it explicitly identified specific information as a rumor. Debunking posts were further labeled as health-related if they concerned human health, including (but not limited to) psychology, diet, medical treatments, disease prevention, lifestyle habits, infectious diseases, or public health events.</p><p>Using the finalized criteria, 3 authors participated in the final annotation. An additional 500 posts were sampled using a fixed random seed, and the first 2 authors independently annotated this set to ensure reproducibility. Interannotator reliability was assessed using Krippendorff &#x03B1;, which accommodates different data types and numbers of coders; values above 0.80 are commonly interpreted as strong reliability [<xref ref-type="bibr" rid="ref44">44</xref>]. After &#x03B1; exceeded 0.80 for both classification tasks, any remaining disagreements were resolved through discussion, with the third author mediating when necessary, to produce the final gold-standard datasets. The resulting &#x03B1; values were 0.835 for debunking identification and 0.909 for health topic classification.</p></sec><sec id="s2-3-3"><title>Model Evaluation and Selection</title><p>To select the best-performing classifier, we evaluated a range of representative approaches in natural language processing using 5-fold cross-validation on the gold-standard datasets. We tested traditional machine learning baselines (support vector machine, random forest, Extreme Gradient Boosting [XGBoost], and logistic regression), a deep learning sequence model (attention-based bidirectional long short-term memory), and transformer-based pretrained language models (bert-base-chinese and Chinese-roberta-wwm-ext). We also evaluated Sentence Transformer Fine-Tuning (SetFit), a label-efficient framework that fine-tunes sentence transformers with relatively few labeled examples [<xref ref-type="bibr" rid="ref45">45</xref>]. For SetFit, we compared multiple backbone models (bert-base-chinese, Chinese-bert-wwm-ext, paraphrase-multilingual-MiniLM-L12-v2, and Chinese-roberta-wwm-ext) to identify the most suitable option for our Chinese corpus, which was subsequently deployed to classify the full candidate dataset.</p></sec></sec><sec id="s2-4"><title>Topic Analysis</title><sec id="s2-4-1"><title>Topic Identification</title><p>To uncover latent themes in health-related rumor debunking posts, we applied BERTopic (BERT: Bidirectional Encoder Representations from Transformers), a topic modeling framework that combines transformer-based sentence embeddings with density-based clustering [<xref ref-type="bibr" rid="ref46">46</xref>]. Unlike probabilistic bag-of-words models such as Latent Dirichlet Allocation, which rely on document-topic distributions, BERTopic captures contextual semantic relationships at the sentence level, making it particularly suitable for short and colloquial Weibo texts [<xref ref-type="bibr" rid="ref47">47</xref>].</p><p>Before topic modeling, we conducted light preprocessing to reduce platform-specific noise while preserving sentence structure for embedding generation. Specifically, we removed topic hashtags enclosed by &#x201C;#&#x201D; (eg, #health-related rumor-debunking#), user mentions starting with &#x201C;@&#x201D; (eg, @People&#x2019;s Daily), and URLs. After cleaning, 2 posts with empty content were excluded.</p><p>BERTopic was then applied to the processed corpus. Texts were embedded using the sentence transformer model paraphrase-multilingual-mpnet-base-v2 [<xref ref-type="bibr" rid="ref48">48</xref>]. We reduced the embedding dimensionality to 5 components using Uniform Manifold Approximation and Projection and clustered the reduced embeddings with Hierarchical Density-Based Spatial Clustering of Applications with Noise (HDBSCAN). HDBSCAN uses a soft-clustering approach that accommodates structural noise by modeling ambiguous texts as outliers (topic &#x2212;1) [<xref ref-type="bibr" rid="ref46">46</xref>]. By preventing the forced assignment of unrelated documents, this mechanism improves topic coherence and semantic interpretability.</p><p>For each cluster, we extracted the 10 most representative terms using class-based Term Frequency&#x2013;Inverse Document Frequency, which estimates term importance across document clusters rather than individual documents, thereby generating robust topic-word distributions [<xref ref-type="bibr" rid="ref46">46</xref>]. To improve interpretability, we filtered topic keywords using a publicly available Chinese stop-word list (Baidu stop-word list) and removed numerical digits (0&#x2010;9) to avoid noninformative tokens in topic descriptions.</p></sec><sec id="s2-4-2"><title>Topic Validation and Labeling</title><p>To verify that the HDBSCAN outlier exclusion (topic &#x2212;1) was temporally unbiased and did not distort the longitudinal dataset, we compared the month-by-month distribution of outlier versus retained posts across the 2009 to 2024 study window using Spearman rank correlation and the Kolmogorov-Smirnov test. Having confirmed the temporal robustness of the dataset, we then quantitatively evaluated the semantic quality of the retained clusters by calculating topic coherence and topic diversity scores based on the top representative keywords.</p><p>While these quantitative metrics confirmed the structural robustness of the clusters, assigning accurate labels to colloquial social media discourse requires deeper semantic comprehension. Therefore, we used a human-in-the-loop framework to finalize the topic representations. For each of the top 10 topics, we extracted the 10 highest-ranked class-based Term Frequency&#x2013;Inverse Document Frequency keywords, along with the 5 most representative documents (selected based on the highest count of matched core keywords, prioritizing longer texts when match scores were equal to ensure maximum contextual richness for the LLM). We then prompted Doubao (Doubao-Seed-2.0-pro-260215), a leading Chinese LLM according to the SuperCLUE [<xref ref-type="bibr" rid="ref49">49</xref>] benchmark, to generate preliminary academic labels using the extracted keywords and documents as context. Finally, one of the authors reviewed each generated label against the raw representative documents and the broader cluster content; any labels deemed potentially inaccurate or ambiguous were discussed and refined through team consensus among the co-authors to ensure semantic accuracy.</p></sec><sec id="s2-4-3"><title>Temporal Trend Analysis</title><p>To examine how health-related rumor debunking themes evolved over the study period, we assigned each post its corresponding BERTopic topic label and aligned these labels with publication time stamps. We then computed and visualized the topic frequency over time, which enabled us to identify shifts in prominence and the emergence of new themes across years.</p></sec></sec><sec id="s2-5"><title>Group Characteristics Analysis</title><sec id="s2-5-1"><title>Automated Information Extraction via LLMs</title><p>Given the finite resources available for large-scale information governance, accurately identifying vulnerable groups is critical for efficiently targeting health-related rumor debunking. To characterize the populations referenced in debunking posts and examine their associations with specific health topics, we used DeepSeek-V3.2, an LLM with strong Chinese language understanding and instruction-following capability [<xref ref-type="bibr" rid="ref50">50</xref>], to build a standardized information extraction pipeline. The model was accessed via an OpenAI-compatible API and applied to the full dataset in batch processing mode. We designed an extraction schema comprising 4 key variables to capture the multidimensional nature of digital health vulnerability: gender and age provide the core demographic strata; social roles (eg, occupation) capture contextual vulnerability beyond biological attributes; and health content reflects the specific medical concerns linked to misinformation.</p><p>To transform diverse textual features into quantifiable demographic variables, we enforced strict standardization through a rule-embedded system prompt. We designed a comprehensive instruction set that directs the model to map extracted features based on predefined criteria. For gender, diverse referents ranging from specific terms such as &#x201C;Mr.&#x201D; to implicit relational terms such as &#x201C;mother&#x201D; are mapped to unified &#x201C;male&#x201D; or &#x201C;female&#x201D; categories. Regarding age, we established a hybrid classification framework anchoring definitions in legal and medical standards. Specifically, minors (aged &#x003C;18 y) and the older adults (aged &#x2265;60 y) follow the strict definitions of the Law on the Protection of Minors [<xref ref-type="bibr" rid="ref51">51</xref>] and the Law on Protection of the Rights and Interests of the Elderly of the People&#x2019;s Republic of China [<xref ref-type="bibr" rid="ref52">52</xref>], respectively. The intermediate spectrum is further stratified into youth (aged 18-44 y) and middle-aged (aged 45-59 y), drawing upon epidemiological evidence regarding the onset of chronic noncommunicable diseases [<xref ref-type="bibr" rid="ref53">53</xref>].</p><p>Furthermore, we addressed the structural complexity of social media content, particularly &#x201C;listicle&#x201D; style posts where a single text often debunks multiple unrelated rumors simultaneously. To resolve this, we integrated structural demonstrations (few-shot examples) into the prompt. This guided the model to parse complex texts and segment distinct semantic units, ensuring that social roles and specific health content were extracted as structured lists corresponding to each subtopic, thereby preserving semantic granularity and completeness.</p></sec><sec id="s2-5-2"><title>Validation of the Extraction Framework</title><p>Before large-scale deployment, we validated the reliability of this extraction framework using a 200-post gold-standard set annotated independently by 2 coders. Interannotator agreement (IAA) was first calculated to establish the reliability of the human annotations. The model&#x2019;s extraction performance was assessed by comparing the outputs of DeepSeek-V3.2 with the human-adjudicated ground truth. Evaluation metrics were selected to match the structural characteristics of each extracted attribute. For standardized categorical variables, specifically the mapped gender and age groups, Krippendorff &#x03B1; was used to measure IAA, and the Macro <italic>F</italic><sub>1</sub>-score was used to evaluate model performance. For unordered entity lists, such as raw demographic mentions and identity labels, a set-based <italic>F</italic><sub>1</sub>-score was applied. Finally, for free-text health narratives, BERTScore was used to assess semantic similarity beyond surface-level lexical overlap. The validation results demonstrated high IAA across all dimensions, and the pipeline achieved strong alignment with the adjudicated benchmark, thereby supporting its robustness for extracting group mentions and health content from the full corpus (n=93,799). Detailed validation metrics, including overall performance evaluations and category-specific metrics for gender and age variables, are provided in Tables S1-S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-5-3"><title>Demographic and Thematic Association Analysis</title><p>On the basis of the standardized outputs returned by the validated pipeline, we conducted descriptive and association analyses to characterize group mentions and their topical distributions. We first calculated the frequency of extracted gender and age groups across the top BERTopic topics. We then performed word frequency analysis on the extracted social roles and health content to summarize salient health narratives for each demographic subgroup using Chinese word segmentation and stop-word filtering. The most frequent keywords were computed to identify prominent social identities and compare differences in health concerns across demographic strata.</p></sec></sec><sec id="s2-6"><title>Prediction and Explanation of Information Diffusion via XGBoost and Shapley Additive Explanations</title><sec id="s2-6-1"><title>Overview</title><p>To examine the complex factors associated with the propagation of health-related rumor debunking posts, it was necessary to account for the nonlinear dynamics and diminishing returns inherent in social media diffusion [<xref ref-type="bibr" rid="ref54">54</xref>]. To capture these patterns while maintaining interpretability, we adopted an interpretable machine learning framework combining XGBoost Regressor for predicting continuous engagement outcomes [<xref ref-type="bibr" rid="ref55">55</xref>], with SHAP (Shapley Additive Explanations) [<xref ref-type="bibr" rid="ref56">56</xref>] for model interpretation. This framework enabled us to estimate the expected magnitude of user engagement while quantifying the marginal contributions of source characteristics, content attributes, and contextual cues, thereby facilitating a more transparent understanding of the mechanisms underlying rumor debunking diffusion [<xref ref-type="bibr" rid="ref57">57</xref>].</p></sec><sec id="s2-6-2"><title>Feature Engineering and Target Variable Transformation</title><p>To mitigate potential omitted-variable bias and capture the multidimensional context of information diffusion on social media platforms, we engineered features across 3 major dimensions: user metadata, content characteristics, and temporal and contextual attributes.</p><p>First, regarding user metadata, we included account verification type and membership category as categorical features, while ordinal membership ranks were label encoded. To reduce skewness and stabilize variance, highly skewed account-level metrics&#x2014;including follower count, following count, and status count&#x2014;were transformed using a base-10 logarithm.</p><p>Second, for content characteristics, we incorporated 4 primary dimensions: thematic categories, multimedia elements, text length, and sentiment polarity. Thematic categories were derived from BERTopic, and multimedia elements were represented by the presence of attached images and videos. Regarding text length, it was captured through 2 distinct features: a log-transformed continuous variable to measure overall length and a binary indicator for posts exceeding 140 characters. This design accounts for Weibo&#x2019;s historical character limits and the platform&#x2019;s current interface mechanism, in which longer posts are truncated and require users to click &#x201C;expand&#x201D; to view the full content&#x2014;an additional interaction step that may introduce friction and influence user engagement behaviors.</p><p>To capture emotional attributes, sentiment polarity for the full corpus (n=93,799) was first classified into 3 categories (positive, neutral, and negative) using the Gemma4:e4b LLM (Google DeepMind) through prompt-based annotation. To evaluate the reliability of this automated annotation, we then constructed a silver-standard reference dataset&#x2014;a high-quality reference set generated primarily through automated multimodel consensus, with human adjudication reserved for cases of full disagreement&#x2014;through stratified sampling: 500 posts were randomly drawn from each of the 3 Gemma4:e4b-classified categories (1500 in total) to ensure adequate evaluation coverage across all sentiment classes, including the typically less frequent ones. These 1500 posts were independently reannotated using identical prompts by 3 Chinese LLMs&#x2014;Doubao (Doubao-Seed-2.0-pro-260215 and ByteDance), Kimi (Kimi-K2.5-Thinking and Moonshot AI), and Qwen (Qwen3.5-397B-A17B-Thinking and Alibaba Cloud)&#x2014;which had no access to the Gemma4:e4b labels. Intermodel consistency among the 3 evaluator models was assessed using the ordinal Krippendorff &#x03B1; coefficient, and final silver-standard labels were determined through majority voting, with human adjudication applied in cases of complete disagreement among the 3 models. Agreement between the Gemma4:e4b annotations and the silver-standard labels was subsequently evaluated using the Quadratically Weighted Cohen &#x03BA; (QWK) coefficient, which accounts for the ordinal structure of sentiment polarity.</p><p>Finally, to capture temporal and contextual characteristics of posting behavior, we engineered categorical features representing posting hour, season, weekend status, and posting device source, which on Weibo is publicly visible and reflects the operational context of the posting behavior (eg, mobile devices vs professional web management clients). To reduce feature sparsity and avoid overfitting to infrequent platforms, only the 10 most common device sources were retained as independent categories, while all remaining long-tail sources were grouped into an &#x201C;Other&#x201D; category.</p><p>Regarding the target variables (reposts, comments, and likes), social media engagement metrics typically exhibit highly skewed long-tail distributions. To stabilize variance and reduce the disproportionate influence of extreme outliers, all engagement counts were transformed using a base-10 logarithmic transformation (log&#x2081;&#x2080;(x+1)). These transformed engagement metrics served as the continuous target variables for the regression models.</p></sec><sec id="s2-6-3"><title>Model Optimization and Validation</title><p>We selected XGBoost as the primary modeling approach because gradient-boosted decision trees are well suited to capturing complex interactions and diminishing returns common in diffusion processes [<xref ref-type="bibr" rid="ref58">58</xref>]. In addition, prior benchmarks on tabular prediction tasks have shown that tree-based ensembles often provide strong accuracy and computational efficiency at this scale [<xref ref-type="bibr" rid="ref59">59</xref>]. As reposts, comments, and likes represent different modes of user engagement, model optimization was conducted independently for each target variable. Using the Optuna framework, we performed independent hyperparameter optimization (eg, tuning learning rate, max depth, and subsample) to minimize the validation root-mean-squared error (RMSE) for each target variable, yielding 3 distinct models. To prevent data leakage from users with multiple posts, these models were evaluated using a rigorous 5-Fold GroupKFold Cross-Validation strategy, grouped by user ID. Model performance was quantified using <italic>R</italic>&#x00B2; and RMSE.</p></sec><sec id="s2-6-4"><title>Interpreting Feature Contributions Using SHAP</title><p>To interpret the fitted regression models, we used SHAP, a game-theoretic framework that decomposes continuous predictions into feature-level contributions [<xref ref-type="bibr" rid="ref56">56</xref>]. SHAP supports 2 complementary forms of explanation in our study: (1) global feature importance, which identifies the most influential predictors across the corpus; and (2) dependence plots, which visualize nonlinear effects and reveal potential thresholds where a predictor&#x2019;s marginal contribution changes sharply [<xref ref-type="bibr" rid="ref57">57</xref>]. Together, these analyses provide an interpretable account of what is associated with engagement magnitude and diffusion in health-related rumor debunking posts.</p></sec></sec><sec id="s2-7"><title>Ethical Considerations</title><p>All data were collected exclusively from publicly accessible posts on the Sina Weibo platform. The study involved retrospective analysis of publicly available content and did not include any direct interaction with or intervention involving human participants. To protect user privacy, analyses were conducted at an aggregate level, and any personally identifiable information was removed or anonymized prior to analysis.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Two-Stage Classification Results</title><p>We constructed the final analytical sample using a 2-stage classification pipeline. In stage 1 (debunking identification), SetFit with the Chinese-roberta-wwm-ext backbone achieved the best performance among 11 candidate models, with a weighted <italic>F</italic><sub>1</sub>-score of 0.8973, indicating strong discriminative performance (<xref ref-type="table" rid="table1">Table 1</xref>). Applying this model to the 413,540 candidate posts identified 377,520 rumor debunking posts.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Performance comparison of classification models for debunking identification (stage 1).</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Accuracy</td><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (macro)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (weighted)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="6">Traditional machine learning</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>SVM<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="top">0.7000</td><td align="left" valign="top">0.8971</td><td align="left" valign="top">0.7262</td><td align="left" valign="top">0.5888</td><td align="left" valign="top">0.7342</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Random forest</td><td align="left" valign="top">0.6100</td><td align="left" valign="top">0.8571</td><td align="left" valign="top">0.6429</td><td align="left" valign="top">0.4994</td><td align="left" valign="top">0.6594</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>XGBoost<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">0.6100</td><td align="left" valign="top">0.8814</td><td align="left" valign="top">0.6190</td><td align="left" valign="top">0.5215</td><td align="left" valign="top">0.6614</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Logistic regression</td><td align="left" valign="top">0.8100</td><td align="left" valign="top">0.8421</td><td align="left" valign="top">0.9524</td><td align="left" valign="top">0.4945</td><td align="left" valign="top">0.7661</td></tr><tr><td align="left" valign="top" colspan="6">Standard deep learning</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Att-BLSTM<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">0.6100</td><td align="left" valign="top">0.9245</td><td align="left" valign="top">0.5833</td><td align="left" valign="top">0.5481</td><td align="left" valign="top">0.6618</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>bert<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup>-base-chinese</td><td align="left" valign="top">0.8800</td><td align="left" valign="top">0.9186</td><td align="left" valign="top">0.9405</td><td align="left" valign="top">0.7647</td><td align="left" valign="top">0.8767</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Chinese-roberta<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup>-wwm-ext</td><td align="left" valign="top">0.9000</td><td align="left" valign="top">0.9302</td><td align="left" valign="top">0.9524</td><td align="left" valign="top">0.8039</td><td align="left" valign="top">0.8973</td></tr><tr><td align="left" valign="top" colspan="6">Few-shot learning</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>SetFit<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup> (Chinese-bert-wwm-ext)</td><td align="left" valign="top">0.8900</td><td align="left" valign="top">0.9195</td><td align="left" valign="top">0.9524</td><td align="left" valign="top">0.7782</td><td align="left" valign="top">0.8853</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>SetFit (bert-base-chinese)</td><td align="left" valign="top">0.8900</td><td align="left" valign="top">0.9101</td><td align="left" valign="top">0.9643</td><td align="left" valign="top">0.7645</td><td align="left" valign="top">0.8814</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>SetFit (paraphrase-multilingual-MiniLM-L12-v2)</td><td align="left" valign="top">0.8900</td><td align="left" valign="top">0.9195</td><td align="left" valign="top">0.9524</td><td align="left" valign="top">0.7782</td><td align="left" valign="top">0.8853</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>SetFit(Chinese-roberta-wwm-ext)<sup><xref ref-type="table-fn" rid="table1fn7">g</xref></sup></td><td align="left" valign="top">0.9000</td><td align="left" valign="top">0.9302</td><td align="left" valign="top">0.9524</td><td align="left" valign="top">0.8039</td><td align="left" valign="top">0.8973</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>SVM: support vector machine.</p></fn><fn id="table1fn2"><p><sup>b</sup>XGBoost: Extreme Gradient Boosting.</p></fn><fn id="table1fn3"><p><sup>c</sup>Att-BLSTM: Attention-based Bidirectional Long Short-Term Memory.</p></fn><fn id="table1fn4"><p><sup>d</sup>BERT: Bidirectional Encoder Representations from Transformers.</p></fn><fn id="table1fn5"><p><sup>e</sup>RoBERTa: a robustly optimized BERT pretraining approach.</p></fn><fn id="table1fn6"><p><sup>f</sup>SetFit: Sentence Transformer Fine-Tuning.</p></fn><fn id="table1fn7"><p><sup>g</sup>Best performing model overall.</p></fn></table-wrap-foot></table-wrap><p>In stage 2 (health topic classification), the same SetFit (Chinese-roberta-wwm-ext) configuration again performed best, reaching a weighted <italic>F</italic><sub>1</sub>-score of 0.9405, reflecting robust classification performance (<xref ref-type="table" rid="table2">Table 2</xref>). When applied to the 377,520 debunking posts, it yielded 93,799 health-related posts (24.8%) and 283,721 non&#x2013;health-related posts (75.2%).</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Performance comparison of classification models for health topic classification (stage 2).</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Accuracy</td><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (macro)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (weighted)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="6">Traditional machine learning</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>SVM<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td><td align="left" valign="top">0.7500</td><td align="left" valign="top">0.6190</td><td align="left" valign="top">0.4333</td><td align="left" valign="top">0.6710</td><td align="left" valign="top">0.7355</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Random forest</td><td align="left" valign="top">0.6600</td><td align="left" valign="top">0.4412</td><td align="left" valign="top">0.5000</td><td align="left" valign="top">0.6094</td><td align="left" valign="top">0.6656</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>XGBoost<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td><td align="left" valign="top">0.6100</td><td align="left" valign="top">0.3953</td><td align="left" valign="top">0.5667</td><td align="left" valign="top">0.5793</td><td align="left" valign="top">0.6248</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Logistic regression</td><td align="left" valign="top">0.7600</td><td align="left" valign="top">0.6500</td><td align="left" valign="top">0.4333</td><td align="left" valign="top">0.6800</td><td align="left" valign="top">0.7440</td></tr><tr><td align="left" valign="top" colspan="6">Standard deep learning</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Att-BLSTM<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td><td align="left" valign="top">0.7000</td><td align="left" valign="top">0.5000</td><td align="left" valign="top">0.2667</td><td align="left" valign="top">0.5765</td><td align="left" valign="top">0.6680</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>bert<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup>-base-chinese</td><td align="left" valign="top">0.9200</td><td align="left" valign="top">0.8667</td><td align="left" valign="top">0.8667</td><td align="left" valign="top">0.9048</td><td align="left" valign="top">0.9200</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Chinese-roberta<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup>-wwm-ext</td><td align="left" valign="top">0.9100</td><td align="left" valign="top">0.8387</td><td align="left" valign="top">0.8667</td><td align="left" valign="top">0.8939</td><td align="left" valign="top">0.9104</td></tr><tr><td align="left" valign="top" colspan="6">Few-shot learning</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>SetFit<sup><xref ref-type="table-fn" rid="table2fn6">f</xref></sup> (Chinese-bert-wwm-ext)</td><td align="left" valign="top">0.9200</td><td align="left" valign="top">0.8929</td><td align="left" valign="top">0.8333</td><td align="left" valign="top">0.9029</td><td align="left" valign="top">0.9192</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>SetFit (bert-base-chinese)</td><td align="left" valign="top">0.9100</td><td align="left" valign="top">0.8889</td><td align="left" valign="top">0.8000</td><td align="left" valign="top">0.8896</td><td align="left" valign="top">0.9086</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>SetFit (paraphrase-multilingual-MiniLM-L12-v2)</td><td align="left" valign="top">0.9400</td><td align="left" valign="top">0.9615</td><td align="left" valign="top">0.8333</td><td align="left" valign="top">0.9256</td><td align="left" valign="top">0.9387</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>SetFit (Chinese-roberta-wwm-ext)<sup><xref ref-type="table-fn" rid="table2fn7">g</xref></sup></td><td align="left" valign="top">0.9400</td><td align="left" valign="top">0.8750</td><td align="left" valign="top">0.9333</td><td align="left" valign="top">0.9299</td><td align="left" valign="top">0.9405</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>SVM: support vector machine.</p></fn><fn id="table2fn2"><p><sup>b</sup>XGBoost: Extreme Gradient Boosting.</p></fn><fn id="table2fn3"><p><sup>c</sup>Att-BLSTM: Attention-based Bidirectional Long Short-Term Memory.</p></fn><fn id="table2fn4"><p><sup>d</sup>BERT: Bidirectional Encoder Representations from Transformers.</p></fn><fn id="table2fn5"><p><sup>e</sup>RoBERTa: a robustly optimized BERT pretraining approach.</p></fn><fn id="table2fn6"><p><sup>f</sup>SetFit: Sentence Transformer Fine-Tuning.</p></fn><fn id="table2fn7"><p><sup>g</sup>Best performing model overall.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Temporal and Source Characteristics of Health-Related Rumor Debunking on Weibo</title><p>Using the health-related debunking posts identified by the 2-stage pipeline, we examined temporal trends and source composition from 2009 to 2024.</p><sec id="s3-2-1"><title>Temporal Patterns of Health-Related Rumor Debunking</title><p>From 2009 to 2024, the overall volume of rumor debunking posts increased substantially, with 2 pronounced peaks in 2020 and 2024 (<xref ref-type="fig" rid="figure2">Figure 2A</xref>). Importantly, these peaks differed in topical composition. The 2020 peak was characterized by an unprecedented concentration of health-related debunking posts, temporally aligned with the outbreak of the COVID-19 pandemic. In contrast, the 2024 peak was largely composed of non&#x2013;health-related topics, indicating that the overall expansion of debunking activity did not translate into a parallel increase in health-related content.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Temporal patterns of rumor debunking posts on Sina Weibo. (A) Annual post volume and the percentage of health-related content (2009&#x2010;2024). (B) Heatmap of the monthly intensity of health-related content (2010&#x2010;2024). Data from 2009 were excluded from the heatmap due to incomplete temporal coverage (available only from August to December).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e91516_fig02.png"/></fig><p>To further clarify how health-related debunking responded to major public health events, we examined monthly variation in the share of health-related posts (<xref ref-type="fig" rid="figure2">Figure 2B</xref>). The proportion rose sharply to 69.6% in January 2020 and 74.2% in February 2020, corresponding to the initial outbreak period. A second major spike occurred in December 2022 (75.3%), coinciding with the nationwide relaxation of COVID-19 containment policies and heightened public concern about infections and medical supplies. Although the largest spikes clustered in winter months, elevated proportions were also observed in other periods (eg, April 2022: 50.4%; May 2021: 52.1%; and May 2011: 66.7%), indicating that intensified health-related debunking was not confined to winter months and is unlikely to be explained by fixed seasonality alone.</p><p>Beyond these event-driven surges, we observed a structural shift in baseline activity following the COVID-19 outbreak. As shown in <xref ref-type="fig" rid="figure2">Figure 2A</xref>, health-related debunking remained consistently higher during 2020 to 2024 than in the prepandemic period (2009 to 2019), indicating sustained activity in this domain throughout the postpandemic period.</p><p>Despite this sustained volume, the relative predominance of health topics did not persist into 2024. Although total rumor debunking activity reached its highest level in that year, the health-related share declined markedly and remained between 8.3% and 12.2% throughout the year. While the absolute volume remained significant, in relative terms, the proportion approached the lower levels observed during the prepandemic period (2014&#x2010;2019). Taken together, these results suggest that while rumor refutation remained highly active overall, the platform&#x2019;s debunking emphasis shifted away from pandemic-centered health topics in 2024.</p></sec><sec id="s3-2-2"><title>Sources of Rumor Debunking Posts</title><p>To characterize the actors involved in health-related rumor debunking, we first examined the verification types of publishers within the health domain (<xref ref-type="table" rid="table3">Table 3</xref>). Government-verified accounts accounted for the largest share of health-related debunking, contributing 43,207 (64.69%) posts. Media accounts ranked second (n=10,126, 15.16%), followed by institutions (n=5632, 8.43%) and enterprises (n=5380, 8.05%). Campus and nonprofit organizations accounted for only a marginal share, together contributing less than 4% of health-related posts.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Comparison of verified user types between health-related and non&#x2013;health-related rumor debunking posts<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>. Posts without valid verification-type labels (n=95,481) were excluded from this analysis.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Verified user type</td><td align="left" valign="bottom">Health-related posts (n=66,796), n (%)</td><td align="left" valign="bottom">Non&#x2013;health-related posts (n=215,243), n (%)</td><td align="left" valign="bottom">Total (N=282,039), n (%)</td></tr></thead><tbody><tr><td align="left" valign="top">Government</td><td align="left" valign="top">43,207 (64.69)</td><td align="left" valign="top">127,824 (59.39)</td><td align="left" valign="top">171,031 (60.64)</td></tr><tr><td align="left" valign="top">Media</td><td align="left" valign="top">10,126 (15.16)</td><td align="left" valign="top">51,391 (23.88)</td><td align="left" valign="top">61,517 (21.81)</td></tr><tr><td align="left" valign="top">Institution</td><td align="left" valign="top">5632 (8.43)</td><td align="left" valign="top">20,039 (9.31)</td><td align="left" valign="top">25,671 (9.10)</td></tr><tr><td align="left" valign="top">Enterprise</td><td align="left" valign="top">5380 (8.05)</td><td align="left" valign="top">14,841 (6.89)</td><td align="left" valign="top">20,221 (7.17)</td></tr><tr><td align="left" valign="top">Campus</td><td align="left" valign="top">2438 (3.65)</td><td align="left" valign="top">1119 (0.52)</td><td align="left" valign="top">3557 (1.26)</td></tr><tr><td align="left" valign="top">Nonprofit</td><td align="left" valign="top">13 (0.02)</td><td align="left" valign="top">29 (0.01)</td><td align="left" valign="top">42 (0.01)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>The distribution of user types differs significantly between health-related and non&#x2013;health-related topics (<italic>&#x03C7;</italic><sup>2</sup><sub>1</sub>=599.5, N=282,039; <italic>P</italic>&#x003C;.001; Cram&#x00E9;r V=0.05).</p></fn></table-wrap-foot></table-wrap><p>To contextualize this source structure, we compared it with the distribution observed in non&#x2013;health-related rumor debunking. Government accounts also constituted the largest contributor outside the health domain (59.39%). Although the overall source hierarchy was similar across domains, government output was more concentrated in health-related debunking: the government-to-media ratio was approximately 4.3 in health-related posts versus 2.5 in non&#x2013;health-related posts.</p><p>We further examined whether the share of government-verified publishers differed between health-related and non&#x2013;health-related rumor debunking posts by conducting a Pearson chi-square test on publisher type (government vs nongovernment) across the 2 domains. The difference was statistically significant (<italic>&#x03C7;</italic>&#x00B2;<sub>1</sub>=599.5, n=282,039; <italic>P</italic>&#x003C;.001); however, the effect size was small (Cram&#x00E9;r <italic>V</italic>=0.05), suggesting that government accounts were only modestly more prevalent in health-related debunking than in non&#x2013;health-related debunking.</p></sec></sec><sec id="s3-3"><title>Topic Modeling and Temporal Patterns</title><sec id="s3-3-1"><title>Topic Modeling Results</title><p>Topic modeling was conducted on 93,797 health-related rumor debunking posts after text preprocessing. BERTopic identified 352 topics. Sensitivity analysis confirmed that the exclusion of outlier documents did not systematically distort temporal patterns: the monthly post counts of outliers and retained posts were highly synchronized (Spearman &#x03C1;=0.95; <italic>P</italic>&#x003C;.001), and their monthly distributions were statistically indistinguishable (Kolmogorov-Smirnov <italic>D</italic>=0.04; <italic>P</italic>=.99). Quantitative evaluation further demonstrated high clustering quality on the retained documents, yielding a topic coherence score of 0.6589 and a topic diversity score of 0.8043. As established measures of semantic consistency [<xref ref-type="bibr" rid="ref60">60</xref>] and topic distinctiveness [<xref ref-type="bibr" rid="ref61">61</xref>], these scores exceed the common empirical reference values of 0.55 and 0.70, respectively, indicating strong semantic consistency and high topic distinctiveness. Building upon these mathematically robust clusters, our human-in-the-loop framework generated the final interpretable labels. <xref ref-type="table" rid="table4">Table 4</xref> reports the 10 most prevalent topics, including post counts, corpus proportions, and representative keywords (translated).</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>The 10 most prevalent topics identified from health-related rumor debunking posts on Sina Weibo<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup>.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Rank</td><td align="left" valign="bottom">Topic</td><td align="left" valign="bottom">Posts, n (%)</td><td align="left" valign="bottom">Top keywords (translated)</td></tr></thead><tbody><tr><td align="left" valign="top">1</td><td align="left" valign="top">Vaccine</td><td align="left" valign="top">1140 (1.22)</td><td align="left" valign="top">Vaccination, vaccine, first, dose, Guangzhou, injection, second, later, no longer, month</td></tr><tr><td align="left" valign="top">2</td><td align="left" valign="top">Debunking report</td><td align="left" valign="top">1031 (1.10)</td><td align="left" valign="top">Zhao (surname), police, according to law, impersonation, administrative detention, epidemic-related, investigation, spread, egregious, reselling</td></tr><tr><td align="left" valign="top">3</td><td align="left" valign="top">Personal care</td><td align="left" valign="top">867 (0.92)</td><td align="left" valign="top">Skin, hair loss, cosmetic facial mask, hair follicles, hair, sunscreen, skin care, silicone oil, shampoo, hair washing</td></tr><tr><td align="left" valign="top">4</td><td align="left" valign="top">Sleep hygiene</td><td align="left" valign="top">733 (0.78)</td><td align="left" valign="top">Sleep, staying up late, nap, falling asleep, exercise, before bed, sleeping well, regular, sleep enough, quality</td></tr><tr><td align="left" valign="top">5</td><td align="left" valign="top">Food myths</td><td align="left" valign="top">721 (0.77)</td><td align="left" valign="top">Food incompatibility, food, 1935, widespread, everything is normal, discard, alarmist, peanut, animal, chestnut</td></tr><tr><td align="left" valign="top">6</td><td align="left" valign="top">PPE<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup> myths</td><td align="left" valign="top">704 (0.75)</td><td align="left" valign="top">Protective mask, pulmonary nodules, ethylene oxide, causing, Ye Xianwei, doubts, Guizhou Province, clarification, long-term, medical department</td></tr><tr><td align="left" valign="top">7</td><td align="left" valign="top">Local misinformation</td><td align="left" valign="top">659 (0.70)</td><td align="left" valign="top">Longnan, Gansu, municipal party committee, cyberspace administration, all-out, online rumors, makeshift hospital, Wuhan, Chengdu Middle School, Shijiazhuang</td></tr><tr><td align="left" valign="top">8</td><td align="left" valign="top">Communicable diseases</td><td align="left" valign="top">656 (0.70)</td><td align="left" valign="top">Novel, coronavirus, all, pneumonia, layered, prevent, antiviral, vinegar fumigation, prevention, salt water</td></tr><tr><td align="left" valign="top">9</td><td align="left" valign="top">Nutrition and lifestyle myths</td><td align="left" valign="top">653 (0.70)</td><td align="left" valign="top">Weight loss, skipping meals, dinner, eating dinner, obesity, carbohydrates, diet, autophagy, staple food, daily</td></tr><tr><td align="left" valign="top">10</td><td align="left" valign="top">Vision and eye health myths</td><td align="left" valign="top">619 (0.66)</td><td align="left" valign="top">Myopia, eyes, National Eye Care Day, prescription, vision, surgery, wearing glasses, presbyopia, glasses, myopia</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Topics are ranked in descending order of post count. Representative keywords were translated from Chinese. Under the current BERTopic settings, 46,616 (49.70%) posts were classified as topic &#x2212;1 by Hierarchical Density-Based Spatial Clustering of Applications with Noise (outliers) and were excluded from topic-specific clusters.</p></fn><fn id="table4fn2"><p><sup>b</sup>PPE: personal protective equipment.</p></fn></table-wrap-foot></table-wrap><p>Overall, topic prevalence was highly dispersed rather than dominated by a small set of themes. Even the most frequent topic (&#x201C;vaccine&#x201D;) accounted for only 1.22% of the corpus, and the remaining top-ranked topics each contributed similarly small shares. To facilitate interpretation, we grouped the top 10 topics into 2 broader domains: public health and crisis-related governance, as well as lifestyle and routine health management.</p><p>The first domain centers on infectious diseases, preventive measures, and crisis communication. For example, &#x201C;vaccine&#x201D; (rank 1) includes terms related to vaccination schedules and dosage, while &#x201C;communicable diseases&#x201D; (rank 8) and &#x201C;personal protective equipment (PPE) myths&#x201D; (rank 6) capture discussions involving pathogens and protective practices. In addition, some clusters, such as &#x201C;debunking report&#x201D; (rank 2) and &#x201C;local misinformation&#x201D; (rank 7), feature governance- and enforcement-related keywords (eg, &#x201C;police,&#x201D; &#x201C;investigation,&#x201D; and place names), suggesting that part of health-related rumor debunking is embedded within broader information governance and administrative enforcement contexts.</p><p>The second domain reflects recurring concerns in everyday health management. &#x201C;Personal care&#x201D; (rank 3) and &#x201C;sleep hygiene&#x201D; (rank 4) relate to appearance- and lifestyle-oriented issues, while diet-related misinformation appears in both &#x201C;food myths&#x201D; (rank 5) and &#x201C;nutrition and lifestyle myths&#x201D; (rank 9). &#x201C;Vision and eye health myths&#x201D; (rank 10) focuses on myopia correction and eye protection. Taken together, these results indicate that health-related rumor debunking on Weibo spans both crisis-oriented public health topics and a wide range of routine concerns about personal well-being.</p></sec><sec id="s3-3-2"><title>Temporal Patterns of Major Topics</title><p><xref ref-type="fig" rid="figure3">Figure 3</xref> presents the temporal trajectories of post volume across the top 10 topics. Topic activity remained relatively low before 2019 but increased markedly thereafter. Overall, the temporal dynamics of these topics fall into 3 archetypes that reflect different attention-mobilization patterns in health-related debunking: event-driven spikes, sustained growth, and recurrent fluctuations.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Temporal patterns of the top 10 health-related rumor debunking topics on Weibo. The time axis begins in 2010 because the earliest post among the top 10 topics was published on March 8, 2010. PPE: personal protective equipment.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e91516_fig03.png"/></fig><p>The first archetype, event-driven spikes (eg, vaccine, food myths, PPE myths, and communicable diseases), is defined by long periods of low activity punctuated by short-lived, high-intensity surges. These topics respond sharply to external shocks and then quickly regress toward baseline levels. For instance, vaccine-related debunking remained marginal before 2020 but rose steeply during the pandemic period, peaking around mid-2021 before declining again. Similarly, PPE myths and communicable diseases exhibit synchronized spikes in early 2020, consistent with the early phase of COVID-19&#x2013;related uncertainty. Food myths differ slightly in that it shows a concentrated, single-pulse peak in 2021 rather than repeated waves.</p><p>In contrast, the second archetype, sustained growth (eg, nutrition and lifestyle myths and vision and eye health myths), shows a gradual upward trajectory with mild fluctuations, rather than abrupt shock-driven surges. These topics increase gradually over time, suggesting an expanding and persistent demand for correction in routine health management. Vision and eye health myths is particularly illustrative, rising from near-zero levels around 2016 and continuing upward to its highest activity in 2024, indicating steadily growing attention in debunking practice.</p><p>The third archetype, recurrent fluctuations (eg, debunking report, personal care, sleep hygiene, and local misinformation), differs from the other 2 patterns in that attention is repeatedly reactivated rather than driven by a single shock or a gradual rise. These topics show irregular volatility with multiple recurring peaks, reflecting episodic bursts of debunking activity over time. Although their overall activity increased after 2019, the repeated rises and declines suggest that attention cycles are triggered intermittently rather than accumulating steadily.</p></sec></sec><sec id="s3-4"><title>Analysis of Extracted Information</title><sec id="s3-4-1"><title>Overall Frequencies of Mentioned Groups</title><p><xref ref-type="table" rid="table5">Table 5</xref> presents the distribution of target audiences identified through post-level targeting feature analysis. The majority of posts were classified as either nonoriented or universally oriented, accounting for 91.31% (n=85,648) of the<underline/> 93,799 posts for gender and 87.64% (n=82,205) for age. This indicates that most health rumor debunking content was not explicitly tailored to specific demographic groups.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Distribution of specific target audiences in health rumors<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup>.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Demographic characteristics</td><td align="left" valign="bottom">Count, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Gender</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Female</td><td align="left" valign="top">5352 (5.71)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Male</td><td align="left" valign="top">2799 (2.98)</td></tr><tr><td align="left" valign="top" colspan="2">Age group</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Minor (&#x003C;18 y)</td><td align="left" valign="top">4878 (5.20)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Youth (18&#x2010;44 y)</td><td align="left" valign="top">5020 (5.35)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Middle aged adults (45&#x2010;59 y)</td><td align="left" valign="top">4058 (4.33)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Older adults (&#x2265;60 y)</td><td align="left" valign="top">3590 (3.83)</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>Age groups were categorized based on legal definitions and epidemiological evidence. Statistics are based on unique posts. For each demographic category, the count represents the number of unique posts containing a specific targeting feature. A single post may mention or target multiple age groups and therefore may be counted in more than one specific age-group category. Posts with no specific demographic features (&#x201C;nonoriented&#x201D;) or with all demographic groups mentioned (&#x201C;universally oriented&#x201D;) were grouped together and excluded from the specific demographic categories.</p></fn></table-wrap-foot></table-wrap><p>Among posts exhibiting explicit demographic orientation, clear disparities were observed. Female-oriented posts (n=5352, 5.71%) substantially outnumbered male-oriented posts (n=2799, 2.98%). With respect to age, youth constituted the most frequently targeted group (n=5020, 5.35%), followed by minors (n=4878, 5.20%) and middle-aged adults (n=4058, 4.33%). In contrast, older adult-oriented posts were the least prevalent among the defined age categories (n=3590, 3.83%).</p></sec><sec id="s3-4-2"><title>Associations Between Health Content and Specific Groups</title><p><xref ref-type="fig" rid="figure4">Figure 4</xref> illustrates the distribution of gender- and age-referenced mentions across the top 10 BERTopic-derived themes in health-related debunking posts. Overall, both demographic dimensions exhibited marked topic-level variation, indicating that references to specific populations were unevenly distributed across thematic contexts.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Frequency of group mentions in different topics. The total frequency of mentions per topic is shown to the right of each bar, while segment labels indicate specific subgroup counts. Labels are omitted for smaller segments for visual clarity. Detailed statistics are provided in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. PPE: personal protective equipment.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e91516_fig04.png"/></fig><p>In terms of gender, the aggregate frequency of mentions across the top 10 topics was relatively balanced between men (n=321) and women (n=312). However, distinct disparities emerged at the thematic level. The topic &#x201C;debunking report&#x201D; accounted for the largest share of gendered references for both genders, with a higher frequency in posts mentioning men (n=292) than women (n=195). In contrast, lifestyle-related topics showed a pronounced skew toward female mentions. &#x201C;Personal care&#x201D; was overwhelmingly associated with women (n=49) compared with men (n=5), and &#x201C;nutrition and lifestyle myths&#x201D; similarly exhibited a female-dominant pattern (women: n=21 and men: n=2). Conversely, men appeared slightly more frequently in technical or acute health topics such as &#x201C;communicable diseases&#x201D; and &#x201C;PPE myths,&#x201D; although the absolute numbers for these categories were low.</p><p>Age-related references displayed a more differentiated thematic distribution. Overall, youth (n=367) and older adults (n=280) were the most frequently mentioned age groups. &#x201C;Vision and eye health myths&#x201D; emerged as the most significant topic in the context of age-specific mentions, accumulating the highest aggregate frequency (n=511). This topic showed broad relevance across the lifespan, ranging from the older adults (n=170) and youth (n=126) to the middle-aged adults (n=113) and minors (n=102), indicating that eye health is a concern shared across age groups rather than confined to any single generation. The &#x201C;youth&#x201D; group dominated the &#x201C;debunking report&#x201D; category (n=116), significantly outpacing other age groups. In addition, &#x201C;personal care&#x201D; was primarily associated with youth (n=33) and minors (n=17), with comparatively few references to the older adults (n=7). The older adult group showed notable representation in &#x201C;sleep hygiene&#x201D; (n=31), ranking second only to youth (n=38).</p><p>Comparing gender- and age-based references reveals differences in how demographic attributes were distributed across topics. &#x201C;Vision and eye health myths&#x201D; accounted for a substantial proportion of age-related mentions but contained relatively few gender-specific references (total gender mentions=15). In contrast, &#x201C;debunking report&#x201D; posts exhibited a high density of gender references (n=487) while containing comparatively fewer age-specific mentions than the vision-related topic. Together, these patterns indicate that age and gender were emphasized in different thematic contexts within health-related debunking posts.</p></sec><sec id="s3-4-3"><title>The Association Between Groups and Specific Topics</title><p>To characterize health narratives associated with different populations, we analyzed the most salient keywords linked to each demographic group (<xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>). Across all strata, pandemic-related terms&#x2014;most notably &#x201C;pandemic,&#x201D; &#x201C;pneumonia,&#x201D; and &#x201C;infection&#x201D;&#x2014;were consistently prominent, establishing a shared thematic baseline for health rumor debunking during the study period. Beyond this common context, distinct gender- and age-specific keyword patterns were observed.</p><p>The female-associated keyword profile was dominated by terms related to reproductive and maternal health, including &#x201C;uterus,&#x201D; &#x201C;ovary,&#x201D; &#x201C;gestational period,&#x201D; &#x201C;fetus,&#x201D; and &#x201C;pregnancy brain.&#x201D; In contrast, the male-associated profile exhibited a markedly different focus, featuring keywords related to acute cardiovascular events and pandemic control measures, such as &#x201C;cardiac arrest,&#x201D; &#x201C;heart,&#x201D; &#x201C;choking,&#x201D; &#x201C;lockdown,&#x201D; and &#x201C;quarantine.&#x201D; Reproductive health&#x2013;related terms were largely absent from the male profile.</p><p>For minors, the dominant keywords clustered around conditions and care-related contexts, with terms such as &#x201C;myopia&#x201D; and &#x201C;leukemia&#x201D; co-occurring with caregiving-related keywords, including &#x201C;injection,&#x201D; &#x201C;taking medicine,&#x201D; and &#x201C;milk.&#x201D;</p><p>The youth group displayed a heterogeneous keyword set combining lifestyle-related behaviors&#x2014;such as &#x201C;cola,&#x201D; &#x201C;eating hotpot,&#x201D; and &#x201C;staying up late&#x201D;&#x2014;with terms denoting severe acute health outcomes, including &#x201C;sudden death,&#x201D; &#x201C;myocardial infarction,&#x201D; and &#x201C;esophageal cancer.&#x201D;</p><p>For the middle-aged group, prominent keywords clustered around obstetric and reproductive events (eg, &#x201C;premature birth,&#x201D; &#x201C;hemorrhage,&#x201D; and &#x201C;gave birth&#x201D;), alongside general health maintenance terms such as &#x201C;soybean&#x201D; and &#x201C;exercise.&#x201D;</p><p>In contrast, the older adult keyword profile was dominated by pandemic-related and prevention-oriented terms, including &#x201C;COVID-19,&#x201D; &#x201C;testing,&#x201D; &#x201C;vaccine,&#x201D; &#x201C;vaccination,&#x201D; &#x201C;pneumonia,&#x201D; &#x201C;virus,&#x201D; and &#x201C;infection.&#x201D; Alongside these, keywords reflecting age-specific vulnerability and support contexts&#x2014;such as &#x201C;heatstroke,&#x201D; &#x201C;decline,&#x201D; &#x201C;function,&#x201D; and &#x201C;medical insurance&#x201D;&#x2014;were also prominent.</p></sec><sec id="s3-4-4"><title>Distribution of Social Roles and Identity Labels</title><p>While gender and age provide a structural overview of demographic targeting, the distribution of social roles reveals more context-specific forms of situational vulnerability (see <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref> for the top 20 most frequent social roles). Social role mentions were distributed across several distinct categories. Generic identity labels were most prevalent, with &#x201C;netizens&#x201D; (n=3484) and &#x201C;citizens&#x201D; (n=3432) appearing most frequently.</p><p>Among specific social identities, &#x201C;pregnant women&#x201D; (n=1482) and &#x201C;students&#x201D; (n=1473) emerged as the most prominent groups, followed by &#x201C;patients&#x201D; (n=1298) and &#x201C;parents&#x201D; (n=1220).</p><p>Professional roles were also prominently represented, with &#x201C;experts&#x201D; (n=2659) ranking third overall and &#x201C;physicians&#x201D; (n=1479) appearing at high frequency. Together, these patterns suggest that, beyond generalized references to the public, health-related narratives are frequently anchored in professional authority as well as roles associated with reproduction, education, and caregiving.</p></sec></sec><sec id="s3-5"><title>Determinants of Diffusion in Health-Related Rumor Debunking Posts</title><sec id="s3-5-1"><title>Validation of Sentiment Feature Extraction</title><p>To examine the potential impact of sentiment on information diffusion, we first validated the reliability of the extracted sentiment features. The internal consistency among the 3 evaluator LLMs used to construct the silver-standard dataset was substantial (ordinal &#x03B1;=0.694; see <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref> for comprehensive agreement metrics, including Fleiss &#x03BA; and pairwise QWK). This confirms a sufficient and reliable level of consensus in evaluating the complex sentiment polarities of health-debunking discourse. Evaluated against this silver standard, the Gemma-4 annotations achieved a macro-<italic>F</italic><sub>1</sub>-score of 0.629, an overall accuracy of 0.639, and a QWK of 0.576. Despite the inherent subjectivity and extreme semantic complexity of Weibo texts, these metrics collectively indicate a reliable and robust level of accuracy for the automated sentiment classification.</p></sec><sec id="s3-5-2"><title>Predictive Performance of the XGBoost Regressors</title><p>The XGBoost regression models showed moderate explanatory power across all 3 engagement dimensions, consistent with the high inherent variability of social media engagement data. Specifically, the reposts model achieved an <italic>R</italic><sup>2</sup> of 0.225 (RMSE=0.344), the comments model yielded an <italic>R</italic><sup>2</sup> of 0.210 (RMSE=0.335), and the likes model obtained an <italic>R</italic><sup>2</sup> of 0.263 (RMSE=0.572). These results indicate that the models captured meaningful structural and contextual signals associated with user interactions.</p></sec><sec id="s3-5-3"><title>Global Feature Importance and Divergent Engagement Associations</title><p><xref ref-type="fig" rid="figure5">Figure 5</xref> presents the SHAP summary plots for reposts, comments, and likes. Across all 3 dimensions, follower count (log-transformed) consistently ranked as the most important feature. The SHAP values indicate a direct positive association: accounts with larger follower counts were associated with higher engagement predictions, whereas accounts with smaller follower counts were associated with lower predictions.</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Shapley Additive Explanations (SHAP) summary plots for feature contributions to engagement predictions: (A) reposts, (B) comments, and (C) likes. Features are ranked by mean absolute SHAP value; each dot represents one post (red=high feature value, blue=low). &#x201C;Extended text&#x201D; is a binary indicator of whether a post exceeds 140 characters (and is thus truncated with a &#x201C;show more&#x201D; link); &#x201C;weekend status&#x201D; indicates whether the post was published on a weekend.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e91516_fig05.png"/></fig><p>Regarding content attributes, text length and image inclusion showed divergent associations depending on the mode of interaction. In the reposts model (<xref ref-type="fig" rid="figure5">Figure 5A</xref>) and comments model (<xref ref-type="fig" rid="figure5">Figure 5B</xref>), these features showed a positive directional association: longer texts and the presence of attached images were associated with positive marginal contributions to the predicted engagement. Conversely, this dynamic inverted in the likes model (<xref ref-type="fig" rid="figure5">Figure 5C</xref>): shorter texts and the absence of images were associated with higher predicted like counts. Notably, in this model, text length and image presence also emerged as the second and third most influential features, respectively, indicating that content-level attributes played a more central predictive role for likes than for the other engagement types.</p><p>Regarding other source characteristics, beyond follower count discussed earlier, the most prominent account metadata predictors differed across the 3 engagement models. In the reposts model (<xref ref-type="fig" rid="figure5">Figure 5A</xref>), identity verification status emerged as the second most important feature overall; its SHAP values exhibited wide dispersion on both sides of the zero axis, indicating that different verification categories were associated with either positive or negative marginal contributions to the prediction. In the same model, status count showed a predominantly negative directional association, with higher status counts corresponding to lower predicted repost magnitude. In the comments model (<xref ref-type="fig" rid="figure5">Figure 5B</xref>), following count emerged as the second most prominent feature overall, showing a positive association with predicted comments. In the likes model (<xref ref-type="fig" rid="figure5">Figure 5C</xref>), in contrast, no single additional source feature stood out as dominant, with the remaining source characteristics collectively showing moderate predictive importance.</p><p>Across all 3 models, temporal variables (season, time of day, weekend status) and text sentiment polarity consistently ranked in the lower tier of the feature hierarchy. Their SHAP values clustered tightly around the zero axis with minimal horizontal dispersion, indicating that sentiment polarity and posting time were associated with limited marginal contributions to final engagement outcomes.</p></sec><sec id="s3-5-4"><title>Nonlinear Marginal Patterns of Key Predictors</title><p>Guided by their prominent roles associated with specific engagement types as identified in the summary plots, 4 continuous features&#x2014;follower count, text length, status count, and following count&#x2014;were selected for detailed nonlinear analysis. As illustrated in <xref ref-type="fig" rid="figure6">Figure 6</xref>, these dependence plots show how the marginal contribution of each feature shifts dynamically across its value distribution.</p><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Shapley Additive Explanations (SHAP) dependence plots showing the nonlinear patterns of 4 key continuous features in XGBoost (Extreme Gradient Boosting) engagement predictions: (A) reposts, (B) comments, and (C) likes. The columns from left to right represent follower count (log-transformed), text length (log-transformed), status count (log-transformed), and following count (log-transformed). For each subplot, the x-axis represents the log-transformed feature value, and the y-axis indicates the corresponding SHAP value. Custom annotations denote the maximum (red arrows) and minimum (green arrows) SHAP values, with the total SHAP value range displayed at the top of each panel. The range is calculated using exact unrounded values, so subtraction of the displayed rounded labels may result in minor discrepancies.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="jmir_v28i1e91516_fig06.png"/></fig><p>Across all 3 engagement dimensions, follower count exhibited a consistent nonlinear pattern characterized by a distinct threshold. For values below approximately 6.0, the SHAP values were predominantly distributed in the negative region with a relatively flat trajectory. Beyond this threshold, the marginal contributions demonstrated a sharp, positive upward trend. Among the 4 analyzed continuous features, follower count exhibited the widest predictive contribution range, particularly in the likes model (SHAP range&#x2248;2.3).</p><p>The dependence plots for text length revealed contrasting nonlinear patterns across the interaction types. In the likes model (<xref ref-type="fig" rid="figure6">Figure 6C</xref>), the highest positive SHAP values were concentrated at the lower end of the distribution, peaking at approximately 1.3 for very short texts. As text length increased, the marginal contribution exhibited a steep downward trajectory, crossing into the negative region and reaching values around &#x2212;0.4. Conversely, the reposts and comments models (<xref ref-type="fig" rid="figure6">Figures 6A and 6B</xref>) displayed a different dynamic. After an initial dip in the shorter text range, the SHAP values transitioned into a positive upward trend. Notably, this positive trajectory in the comments model was not sustained; the main cluster of SHAP values began to decline after reaching a localized concentration in a mid-length range.</p><p>Status count demonstrated a nonlinear trajectory. At lower levels, the marginal contributions were predominantly positive. As the post count increased, these values declined into the negative region to form a distinct trough, before exhibiting a rebound trend at the higher end of the distribution. Notably, the onset of this downward shift occurred earlier in the likes model compared to reposts and comments.</p><p>Finally, following count exhibited a clear threshold pattern. At the lower end of the distribution, the SHAP values were predominantly negative. However, as the values surpassed a mid-range threshold (approximately 2.0-2.5 on the logarithmic scale), the scatter plots crossed the zero axis and transitioned into a positive trajectory for larger following counts.</p></sec></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Results</title><sec id="s4-1-1"><title>Overview</title><p>By analyzing a comprehensive dataset spanning 15 years of health-related rumor debunking posts on Sina Weibo, this study characterizes the longitudinal evolution, thematic diversity, demographic representations, and engagement predictors of corrective communication in China&#x2019;s digital health ecosystem. Our findings demonstrate that health-related rumor debunking on Weibo was strongly influenced by major public health crises, particularly during the COVID-19 pandemic. At the same time, the topical landscape has progressively diversified over the past decade, with lifestyle-related and well-being&#x2013;related themes showing relatively stable long-term growth. In addition, although most debunking posts targeted the general public, clear differences emerged across gender and age groups in both representation and associated health concerns, and these demographic references were unevenly linked to specific health topics. Finally, interpretable machine learning indicated that engagement was associated primarily with source-level reach, with follower count being the strongest predictor across reposts, comments, and likes, while the contribution of content attributes varied by interaction type and was most pronounced for likes; sentiment and posting time showed limited associations. Considered jointly, these findings reveal a structural asymmetry in the debunking ecosystem that single-dimension analyses would not capture: the content production side is maturing, as thematic coverage is broadening and demographic differentiation is emerging, while the distribution side remains governed by source-level reach largely irrespective of content attributes. This content distribution misalignment means that advances in what is corrected and for whom do not automatically translate into proportional communicative impact.</p></sec><sec id="s4-1-2"><title>Crisis-Driven Evolution and Subsequent Attention Rebalancing</title><p>Our longitudinal findings (2009&#x2010;2024) suggest that the COVID-19 pandemic was associated with substantial shifts in the volume of health rumor debunking on Sina Weibo, with effects that appear to have extended well beyond the acute outbreak period. The sharp escalation of health-related debunking activity in 2020 likely reflected the convergence of 2 parallel dynamics. On the one hand, the outbreak generated unprecedented levels of misinformation under conditions of high public uncertainty, concerning transmission routes, protective measures, vaccines, and treatment strategies. On the other hand, the pandemic simultaneously triggered an intensified institutional response, as government agencies and platforms rapidly expanded corrective communication efforts to stabilize the online information environment during a large-scale public health emergency.</p><p>Beyond this initial surge, the elevated level of health-related debunking activity persisted well past the most acute outbreak stages. Even after the initial outbreak phase (2020) subsided, the volume of health-related debunking remained substantially elevated throughout 2020 to 2023 compared with the prepandemic baseline. This pattern suggests that the pandemic may have functioned not only as a short-term focusing event but also as a catalyst for broader governance reconfiguration. Institutional mechanisms initially mobilized under emergency conditions&#x2014;including specialized debunking accounts, cross-platform coordination, and intensified administrative intervention&#x2014;appear to have become partially embedded within routine governance practices. This pattern can be interpreted through the broader lens of Punctuated Equilibrium Theory [<xref ref-type="bibr" rid="ref62">62</xref>], which proposes that exogenous shocks can disrupt previously stable policy equilibria and generate durable shifts in governance arrangements. However, this analysis tracks volumes of debunking posts rather than the institutional arrangements themselves; therefore, this interpretation should be regarded as suggestive rather than conclusive.</p><p>At the same time, this sustained postpandemic elevation should not be interpreted solely as evidence of continuously heightened public concern regarding health misinformation. The persistence of debunking activity may also reflect the temporary continuation of institutional mandates and governance infrastructures established during the emergency phase. In other words, the prolonged visibility of health rumor governance likely emerged through the interaction between enduring public sensitivity and sustained administrative mobilization.</p><p>By 2024, however, the relative share of health-related debunking declined markedly, despite the overall volume of rumor refutation posts reaching its historical peak. Importantly, this decline does not necessarily indicate a weakening of misinformation governance capacity. Rather, it may reflect a redistribution of institutional attention and potentially of broader public salience as pandemic-related risks gradually lost their exceptional status and competing social issues regained visibility within the broader governance agenda. Alternative explanations&#x2014;including shifts in official governance mandates and potential changes in platform-level recommendation algorithms that may influence content visibility&#x2014;also warrant consideration, although the present dataset does not allow us to formally separate these contributing factors. This mechanism is broadly consistent with the issue attention cycle [<xref ref-type="bibr" rid="ref63">63</xref>], which suggests that highly salient public problems gradually lose centrality as crisis urgency fades and attention shifts toward emerging concerns. As with Punctuated Equilibrium Theory mentioned earlier, this framework is offered as an interpretive lens rather than a directly tested mechanism, given that our data capture debunking activity rather than the underlying processes of public attention themselves.</p><p>Together, these findings outline a dynamic cycle of crisis-driven rumor governance characterized by emergency escalation, partial institutionalization, and subsequent rebalancing of attention across issue domains.</p></sec><sec id="s4-1-3"><title>State-Centered Governance in Health Rumor Debunking</title><p>Our findings reveal a predominantly state-centered configuration in China&#x2019;s health rumor debunking ecosystem. Government-verified accounts contributed nearly two-thirds (64.69%) of all health-related debunking posts, substantially exceeding the combined output of media, institutional, and enterprise accounts. While the share of government accounts was statistically higher in health-related debunking than in non&#x2013;health-related debunking (chi-square test, <italic>P</italic>&#x003C;.001), the effect size was small (Cram&#x00E9;r <italic>V</italic>=0.05), indicating that government dominance is a general feature of Sina Weibo&#x2019;s debunking ecosystem rather than a phenomenon unique to the health domain. Nonetheless, this dominance is amplified within the health context: the government-to-media output ratio reached approximately 4.3:1 in health-related debunking, compared with 2.5:1 in non&#x2013;health-related debunking. This contrast suggests that, although state actors play a leading role across topic domains, their relative weight is even more pronounced when the corrective communication concerns health.</p><p>This configuration is consistent with Media System Dependency Theory [<xref ref-type="bibr" rid="ref64">64</xref>], which posits that audience reliance on particular information sources intensifies under conditions of uncertainty and perceived risk. As health rumors often involve scientific complexity and potential collective consequences&#x2014;such as panic buying, vaccine hesitancy, or heightened public anxiety&#x2014;commercially oriented media, shaped by market incentives and attention dynamics, may be less inclined to function as definitive arbiters of truth. State actors, in contrast, are institutionally positioned to supply administrative authority and offer authoritative reference points that the public can draw upon when navigating conflicting health claims online.</p><p>This state-centered pattern, however, should not be read as a universal template for misinformation governance. Comparative evidence from Canada and the United States points to a more distributed, multiactor model, in which corrective communication is shared across public health agencies (eg, the Centers for Disease Control and Prevention and Health Canada), independent fact-checking organizations, professional medical associations, and academic experts, rather than concentrated within central government accounts [<xref ref-type="bibr" rid="ref65">65</xref>,<xref ref-type="bibr" rid="ref66">66</xref>]. Two factors may help explain this divergence. First, in contexts where institutional trust is fragmented along political or ideological lines, corrective messages delivered through a diverse set of nongovernmental and quasi-governmental actors may achieve broader credibility than messages issued directly by central authorities [<xref ref-type="bibr" rid="ref67">67</xref>,<xref ref-type="bibr" rid="ref68">68</xref>]. Second, in such settings, direct government involvement in online information governance can itself become politically contested, which makes distributed governance both a normative preference and a pragmatic strategy [<xref ref-type="bibr" rid="ref69">69</xref>]. Overall, these contrasts indicate that effective debunking architectures are shaped by broader institutional and sociopolitical contexts and that the Chinese pattern documented here reflects one viable configuration rather than a generalizable blueprint.</p></sec><sec id="s4-1-4"><title>Thematic Evolution: From Acute Survival Concerns to Everyday Well-Being</title><p>The thematic evolution of health rumor debunking suggests a gradual broadening of public attention from acute crisis-related concerns toward a wider range of everyday health and well-being issues. This shift plausibly reflects the increasing ubiquity of social media and broader changes in how health information is accessed, circulated, and interpreted online [<xref ref-type="bibr" rid="ref70">70</xref>]. Within this maturing digital environment, we observe a clear divergence in temporal trajectories across topic types. Event-driven topics, such as vaccines and personal protective equipment, tend to follow an &#x201C;explosive but transient&#x201D; pattern, with sharp spikes closely aligned with major epidemic phases. In contrast, lifestyle-oriented topics, most notably vision and eye health, as well as nutrition, exhibit more sustained growth over time, suggesting enduring relevance beyond discrete crisis periods.</p><p>This observed reorientation of public attention may be interpreted through the lens of Maslow&#x2019;s hierarchy of needs [<xref ref-type="bibr" rid="ref71">71</xref>]. During the height of the pandemic, public attention was disproportionately concentrated on safety-related concerns, including infection avoidance and immediate risk reduction. As the perceived severity of the threat declined, attention appears to have gradually expanded toward issues associated with everyday functioning and quality-of-life management.</p><p>At the same time, the sustained prominence of lifestyle-related rumors can be understood in relation to the &#x201C;medicalization of daily life&#x201D; [<xref ref-type="bibr" rid="ref72">72</xref>], whereby ordinary behaviors&#x2014;such as screen use, diet, and sleep&#x2014;are increasingly framed in medicalized terms that invite expert-like intervention. Within this context, routine health anxieties may accumulate over time, generating ongoing demand for ostensibly &#x201C;scientific&#x201D; optimization and creating a fertile environment for pseudoscientific claims, dietary myths, and &#x201C;quick-fix&#x201D; narratives. The coexistence of these distinct temporal archetypes suggests that effective debunking calls for simultaneously managing acute crisis response and sustained routine correction, a duality whose practical implications are further shaped by how corrections circulate and reach their audiences.</p></sec><sec id="s4-1-5"><title>Demographic Differentiation in Health Narratives: Patterns Across Gender, Age, and Social Roles</title><p>Our demographic analysis suggests that digital health narratives are not neutral but systematically differentiated along gender and age lines. To interpret these disparities, we draw on several sociopsychological frameworks while remaining attentive to the boundaries of our data: because our dataset captures keyword-level textual representations rather than direct psychometric or behavioral measures, we frame these theoretical insights as exploratory interpretations rather than definitive causal mechanisms. In parallel, we situate these narrative differences within the structural realities of the digital environment&#x2014;platform demographics, algorithmic biases, and contemporary digital culture.</p><p>With respect to gender, female-related mentions (n=5352, 5.71%) substantially outnumbered those associated with men (n=2799, 2.98%), and the female-associated keyword profile was dominated by terms related to reproductive and maternal health (eg, &#x201C;uterus,&#x201D; &#x201C;ovary,&#x201D; and &#x201C;pregnancy&#x201D;). This patterned framing could tentatively be interpreted through the lens of Objectification Theory [<xref ref-type="bibr" rid="ref73">73</xref>], which highlights how women&#x2019;s bodies are often rendered salient through appearance- and function-oriented discourses. A more direct alternative explanation lies in the engagement-driven algorithmic structure of social media. Prior research consistently demonstrates that women are more active seekers and sharers of online health information, particularly regarding maternal and routine well-being [<xref ref-type="bibr" rid="ref74">74</xref>-<xref ref-type="bibr" rid="ref76">76</xref>]. Drawing on this prior evidence, one possibility is that platform algorithms&#x2014;which prioritize content with high interaction rates&#x2014;disproportionately amplify reproductive and appearance-oriented debunking content because these topics are documented to generate engagement from female demographics. We note that this remains an inference drawn from external literature rather than from engagement metrics within our own dataset.</p><p>In contrast, male-related keywords clustered around 2 distinct categories: acute and high-severity health events (eg, &#x201C;cardiac arrest&#x201D; and &#x201C;choking&#x201D;) and pandemic control terms (eg, &#x201C;lockdown&#x201D; and &#x201C;quarantine&#x201D;). While the acute-event cluster resonates with tentative insights from Hegemonic Masculinity theory [<xref ref-type="bibr" rid="ref77">77</xref>]&#x2014;which has been used to explain the relative invisibility of routine male health concerns&#x2014;it may also reflect the interaction between gendered health information behaviors and platform visibility dynamics. Prior research suggests that men generally engage less with routine or preventive health information online [<xref ref-type="bibr" rid="ref78">78</xref>]. Drawing on prior platform studies, low-interaction topics may be less likely to receive algorithmic amplification [<xref ref-type="bibr" rid="ref79">79</xref>], whereas acute and emotionally arousing medical emergencies are more commonly associated with rapid sharing and public attention [<xref ref-type="bibr" rid="ref5">5</xref>]. Within this framework, routine male health concerns may remain relatively less visible while acute medical events become more prominent in rumor debunking discussions. The co-occurrence of pandemic control terms in the male-associated profile is plausibly a separate phenomenon, reflecting the broader prominence of these topics during the COVID-19 period rather than a gender-specific pattern. As mentioned earlier, these interpretations rest on external evidence rather than on engagement data within our own corpus.</p><p>Across the life course, vulnerability profiles further diverge. Youth emerged as the most frequently referenced group (n=5020, 5.35%), a pattern that closely mirrors Weibo&#x2019;s user demographics, where young adults constitute the platform&#x2019;s core population [<xref ref-type="bibr" rid="ref80">80</xref>]. Beyond frequency, the youth-associated keyword profile was notably heterogeneous, including both lifestyle-related behaviors&#x2014;such as &#x201C;cola,&#x201D; &#x201C;hotpot,&#x201D; and &#x201C;staying up late&#x201D;&#x2014;and terms denoting severe acute health outcomes, such as &#x201C;sudden death,&#x201D; &#x201C;myocardial infarction,&#x201D; and &#x201C;esophageal cancer.&#x201D; This heterogeneous profile likely reflects the breadth of health concerns relevant to Weibo&#x2019;s predominantly young user base, encompassing both everyday lifestyle topics and the acute medical incidents that periodically attract public attention in this demographic.</p><p>Minors also appeared as a highly visible group (n=4878, 5.20%), reflecting their socially recognized vulnerability and the prioritization of child protection in public discourse. The prominence of caregiving-related keywords (eg, &#x201C;injection,&#x201D; &#x201C;taking medicine,&#x201D; and &#x201C;milk&#x201D;) alongside the high frequency of the &#x201C;parents&#x201D; social role (n=1220) suggests that minors&#x2019; health concerns in this corpus are frequently mediated through caregiver discourse rather than appearing as self-expressed concerns. In line with prior research [<xref ref-type="bibr" rid="ref81">81</xref>], this pattern may reflect parental proxy health information seeking, in which caregivers actively engage with health-related content on behalf of children&#x2014;although our keyword-level data describe textual co-occurrence rather than directly capturing information-seeking behaviors. Correspondingly, the discourse is heavily concentrated on &#x201C;vision and eye health&#x201D; (<xref ref-type="fig" rid="figure4">Figure 4</xref>) and keywords such as &#x201C;myopia&#x201D; and &#x201C;leukemia&#x201D; (<xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>), reflecting both routine developmental concerns and heightened anxieties surrounding severe pediatric conditions. This pattern is further reinforced by the prominence of the &#x201C;student&#x201D; social role (n=1473), suggesting that health narratives about minors are frequently framed within educational settings rather than purely biological trajectories of development.</p><p>Moving further along the life course, the middle-aged group exhibits a distinctive and multifaceted keyword profile, combining high-risk obstetric terms (eg, &#x201C;premature birth,&#x201D; &#x201C;hemorrhage,&#x201D; and &#x201C;gave birth&#x201D;) with general health maintenance terms (eg, &#x201C;soybean&#x201D; and &#x201C;exercise&#x201D;). The presence of obstetric terms potentially indicates that reproductive risk remains a salient concern in middle-aged discourse, a pattern that may be situated within broader demographic shifts in contemporary China, including the growing prevalence of pregnancies at advanced maternal age following the relaxation of birth control policies under the 2- and 3-child initiatives. We note, however, that the present cross-sectional data do not directly link this keyword pattern to those policy shifts. The co-occurrence of health maintenance terms further suggests that middle-aged health discourse is not exclusively framed around acute reproductive risks but also encompasses general well-being concerns.</p><p>By contrast, the older adult group (n=3590) exhibited the lowest overall frequency among the defined age groups, although the gap relative to the other groups was modest. Rather than reflecting outright digital marginalization, this lower frequency may partly arise because older adults are often represented by others rather than through direct self-expression&#x2014;a pattern that, although different in form, parallels the proxy-mediated representation observed for minors above, and is consistent with prior research on age-related disparities in social media use [<xref ref-type="bibr" rid="ref75">75</xref>]. The older adult&#x2013;associated keyword profile was dominated by pandemic-related terms (eg, &#x201C;COVID-19,&#x201D; &#x201C;vaccine,&#x201D; &#x201C;pneumonia,&#x201D; and &#x201C;infection&#x201D;), reflecting the prominence of older adults as a high-risk population during the COVID-19 period. Alongside these, keywords such as &#x201C;heatstroke,&#x201D; &#x201C;decline,&#x201D; &#x201C;function,&#x201D; and &#x201C;medical insurance&#x201D; suggest that older adult health narratives in this corpus also tend to be framed around environmental risks and institutional support, although our data cannot directly speak to the underlying agency dynamics.</p><p>In parallel with these demographic patterns, the high frequency of &#x201C;experts&#x201D; (n=2659) and &#x201C;physicians&#x201D; (n=1479) in the overall corpus points to the co-occurrence of professional roles alongside demographic identity labels within rumor debunking discourse. Although our keyword-level data do not directly distinguish the discursive functions these mentions serve, this co-occurrence pattern is broadly consistent with prior observations of expert-driven health communication, in which professional voices and lay subjects of health concern jointly populate corrective discourse.</p><p>Although the majority of health rumor debunking posts in our corpus address a general audience without explicitly mentioning specific groups, the distinct patterns observed within the targeted subset suggest the potential value of precision communication. For health communication practitioners, these differentiated findings indicate that governance efforts might be optimized by supplementing standard universal broadcasts with tailored messages. When addressing niche health concerns, configuring the framing to the distinct contexts of specific genders, age groups, and social roles could enhance the relevance of corrective information. Across these patterns, we emphasize that our analysis describes systematic textual co-occurrences within rumor debunking discourse and that the underlying mechanisms&#x2014;whether psychological, cultural, or algorithmic&#x2014;warrant direct empirical testing in future research.</p></sec><sec id="s4-1-6"><title>Predictors of Engagement: Source Hierarchy and Differentiated Engagement Modes</title><p>Our engagement analysis suggests a partial &#x201C;source-over-content&#x201D; tendency, particularly for diffusion (reposts) and interactive engagement (comments): the spread of debunking messages appears to be more strongly associated with who delivers the message than with what the message is about. Across all 3 engagement types, follower count (operationalized as the log-transformed follower count) consistently emerged as the dominant predictor, while sentiment and posting time showed comparatively limited predictive contributions. One plausible interpretation is that, in information-saturated social media environments, users often lack the time, cognitive resources, or domain-specific expertise required to systematically evaluate complex health information. Under such conditions, they may rely on readily available peripheral cues&#x2014;such as audience size and verification-related signals&#x2014;as heuristic indicators of credibility and relevance. This behavioral pattern is consistent with the Elaboration Likelihood Model, which describes the role of peripheral processing when motivation or ability for deep evaluation is limited [<xref ref-type="bibr" rid="ref82">82</xref>]. We note, however, that our data permit only predictive associations and cannot directly test the underlying cognitive mechanisms; the Elaboration Likelihood Model is therefore offered as one plausible interpretive lens rather than a tested explanation.</p><p>Crucially, the associations of source and content features diverged depending on the mode of engagement. For reposts, follower count was followed by institutional verification (verified type) as the next most prominent source predictor, suggesting that diffusion breadth is more strongly associated with high-status, institutionally credentialed sources than with the substantive content of the post. For comments, in contrast, following count emerged as the second most prominent source predictor and was positively associated with predicted comments. One possible interpretation is that accounts following many others tend to be more active social participants, whose posts may correspond to environments more conducive to dialogue. Social exchange theory [<xref ref-type="bibr" rid="ref83">83</xref>] offers a complementary interpretive frame in which reciprocal networking signals may correspond to lower psychological thresholds for audience response [<xref ref-type="bibr" rid="ref84">84</xref>], but this interpretation should be treated as exploratory given the observational nature of our data.</p><p>The most striking departure from this source-dominant pattern appeared in the likes model. Although follower count remained the leading predictor, text length and image inclusion rose to become the second and third most prominent features, with shorter texts and the absence of images associated with higher predicted likes. This pattern suggests that the lightweight, low-effort nature of liking may render lightweight content (brief, text-only posts) more compatible with the cognitive and behavioral cost of the engagement act itself. For likes, therefore, content-level attributes appear to play a more central predictive role than for reposts or comments, partially qualifying the broader &#x201C;source-over-content&#x201D; framing.</p><p>Despite these mode-specific nuances, the overarching ecosystem remains marked by a pronounced concentration pattern reminiscent of the Matthew Effect [<xref ref-type="bibr" rid="ref85">85</xref>]. The consistent dominance of follower count across all 3 engagement metrics, together with the predominantly negative SHAP values observed for low-follower accounts, suggests a cold-start dynamic in which accounts lacking sufficient initial audience size struggle to gain visibility. Importantly, this pattern need not reflect user-level cognitive heuristics alone; it may also be shaped substantially by platform-level algorithmic mechanisms. On Weibo, recommendation and ranking algorithms tend to amplify content from accounts that have already accumulated engagement, producing a feedback loop in which visibility begets further visibility. The &#x201C;source-over-content&#x201D; pattern we observe in the data is therefore likely a joint product of user heuristics and platform algorithmic design, rather than a pure reflection of either mechanism alone. Without access to platform algorithmic data, we cannot fully disentangle these 2 contributing pathways, and we offer this interpretation as a hypothesis warranting further investigation.</p><p>Together, these patterns suggest a visibility disadvantage for credible but less-followed sources. As the SHAP dependence plots show, accounts with follower counts below approximately 10&#x2076; were associated with predominantly negative marginal contributions across all 3 engagement dimensions, with a sharp positive inflection emerging only beyond this threshold. Credible but low-resource voices may therefore remain structurally below diffusion thresholds regardless of content quality, implying that effective governance requires not only producing accurate corrections but also enabling their circulation. Platform-level interventions, such as algorithmic amplification mechanisms or endorsement features that boost the visibility of credible grassroots sources, may help ensure that evidence-based correction is not structurally suppressed by deficits in social capital.</p></sec></sec><sec id="s4-2"><title>Limitations</title><p>First, contextual generalizability is constrained by the sociopolitical setting of this study. Our analysis is based on Sina Weibo within China&#x2019;s specific information governance framework. While this context offers a valuable lens into a relatively state-centered debunking ecology, it may limit the transferability of our findings to other institutional environments. In particular, the administrative-led configuration observed here differs from the more distributed, multiactor models discussed earlier, limiting the extent to which our findings generalize to such settings. Future comparative research across political and regulatory systems is needed to examine how governance structures shape the organization and effectiveness of rumor debunking.</p><p>Second, platform specificity may restrict the applicability of our conclusions beyond Weibo&#x2019;s media environment. Sina Weibo functions largely as a text-based public sphere with repost-centered diffusion dynamics. In contrast, emerging platforms such as Douyin (the Chinese counterpart of TikTok) operate with different content formats and recommendation architectures, while WeChat (Tencent) represents a more closed, private-domain communication system. Future work should adopt cross-platform designs to evaluate how media format, network structure, and algorithmic logic jointly influence the visibility and diffusion of debunking content.</p><p>Third, the corpus may be incomplete because of our reliance on keyword-based retrieval at the construction stage. Specifically, using &#x201C;rumor debunking&#x201D; (&#x8F9F;&#x8C23;) as the primary query term may omit corrective messages that perform fact-checking implicitly without explicit debunking labels (eg, physicians providing corrective explanations in routine health guidance). Future research could incorporate semantic dense retrieval methods or iterative active learning pipelines to capture a broader spectrum of implicit corrective communication and improve coverage of grassroots correction practices.</p><p>Fourth, our engagement data are observational, and the SHAP attributions therefore characterize predictive contributions rather than causal effects. Although our models incorporated a comprehensive set of source, content, and contextual features, unobserved factors&#x2014;such as platform-level recommendation weights or offline amplification events&#x2014;may still shape diffusion dynamics. Future studies should use experimental designs or formal causal inference techniques to directly isolate the mechanisms underlying user engagement.</p></sec><sec id="s4-3"><title>Conclusions</title><p>By analyzing 15 years of health-related rumor debunking posts on Sina Weibo (2009&#x2010;2024), this study offers a longitudinal account of official corrective communication along 3 dimensions: its thematic evolution, the populations it references, and the factors associated with its diffusion. First, the focus of health-related debunking broadened over time, moving from acute crises&#x2014;most notably COVID-19&#x2014;toward a wider range of routine health and well-being concerns. Second, most debunking posts were not directed at any specific demographic group; among the minority that explicitly referenced particular populations, references were unevenly distributed across gender and age and were linked to distinct health concerns. Third, our predictive analysis indicates that engagement was associated primarily with source-level reach: follower count was the strongest predictor of reposts, comments, and likes alike, outweighing the substance of the correction. Beyond this shared pattern, the secondary predictors differed across the 3 interaction types, indicating that reposts, comments, and likes are shaped by partly distinct configurations of source and content attributes rather than by a single uniform mechanism.</p><p>Viewed jointly, these 3 dimensions reveal a structural asymmetry in the debunking ecosystem: the content side is maturing in both thematic scope and demographic differentiation, while distribution remains governed by source-level reach largely irrespective of content attributes. This asymmetry carries specific governance implications. First, the coexistence of event-driven spikes and sustained growth trajectories among health-related debunking topics indicates that crisis response capacity alone is insufficient; building long-term, routine communication capacity for lifestyle-related concerns that are steadily gaining prominence may be equally important. Second, the systematic associations between demographic groups and specific health concerns, such as reproductive health terms in female-associated content and pandemic-related terms in older adult-associated content, provide an empirical baseline that practitioners can draw upon to assess the coverage and priorities of current debunking efforts. Third, the SHAP analysis revealed a clear threshold effect for follower count, below which accounts were associated with predominantly negative marginal contributions to engagement, suggesting that credible but low-resource voices face a structural diffusion barrier; platform-level mechanisms that help such sources circulate corrections may therefore be an essential complement to relying on established authoritative accounts alone.</p></sec></sec></body><back><ack><p>The authors declare that generative AI tools were used during the preparation of this manuscript under full human oversight. Specifically, Gemini and DeepSeek were used to assist with code generation and debugging for data analysis, and DeepSeek was additionally used to extract demographic mentions and health-related content from the corpus. Doubao was used to generate preliminary topic labels, and Doubao, Kimi, Qwen, and Gemma-4 (the gemma4:e4b model) were used to conduct automated sentiment classification. Gemini and ChatGPT were also used to support literature searching. In addition, Grammarly, together with Gemini, Claude, and ChatGPT, was used for grammatical and lexical refinement to enhance textual clarity. All AI-assisted outputs were checked for accuracy and appropriateness by the authors before being incorporated into the manuscript. The authors are accountable for the submitted manuscript in its entirety.</p></ack><notes><sec><title>Funding</title><p>This study was supported by the Sichuan Science and Technology Program (grant 2024NSFSC1075).</p></sec><sec><title>Data Availability</title><p>The datasets generated and analyzed during this study are available from the corresponding author on reasonable request.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: YF (lead), XT (supporting)</p><p>Data curation: YF</p><p>Formal analysis: YF (lead), CH (supporting)</p><p>Funding acquisition: XT</p><p>Investigation: YF (lead), CH (supporting), YH (supporting)</p><p>Methodology: YF (lead), CH (supporting), XT (supporting), YH (supporting)</p><p>Project Administration: XT</p><p>Resources: XT</p><p>Software: YF (lead), CH (supporting), XT (supporting).</p><p>Supervision: XT</p><p>Validation: YF (lead), CH (supporting)</p><p>Visualization: YF (lead), YH (supporting), CH (supporting)</p><p>Writing &#x2013; original draft: YF (lead), CH (supporting), XT (supporting), YH (supporting)</p><p>Writing &#x2013; review &#x0026; editing: YF (lead), YH (supporting), XT (supporting), CH (supporting)</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">BERT</term><def><p>Bidirectional Encoder Representations from Transformers</p></def></def-item><def-item><term id="abb2">HDBSCAN</term><def><p>Hierarchical Density-Based Spatial Clustering of Applications with Noise</p></def></def-item><def-item><term id="abb3">IAA</term><def><p>interannotator agreement</p></def></def-item><def-item><term id="abb4">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb5">PPE</term><def><p>personal protective equipment</p></def></def-item><def-item><term id="abb6">QWK</term><def><p>Quadratically Weighted Cohen &#x03BA;</p></def></def-item><def-item><term id="abb7">RMSE</term><def><p>root-mean-squared error</p></def></def-item><def-item><term id="abb8">RQ</term><def><p>research question</p></def></def-item><def-item><term id="abb9">SetFit</term><def><p>Sentence Transformer Fine-tuning</p></def></def-item><def-item><term id="abb10">SHAP</term><def><p>Shapley Additive Explanations</p></def></def-item><def-item><term id="abb11">XGBoost</term><def><p>Extreme Gradient Boosting</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kaplan</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Haenlein</surname><given-names>M</given-names> </name></person-group><article-title>Users of the world, unite! the challenges and opportunities of social media</article-title><source>Bus Horiz</source><year>2010</year><month>01</month><volume>53</volume><issue>1</issue><fpage>59</fpage><lpage>68</lpage><pub-id pub-id-type="doi">10.1016/j.bushor.2009.09.003</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Bakshy</surname><given-names>E</given-names> </name><name name-style="western"><surname>Rosenn</surname><given-names>I</given-names> </name><name name-style="western"><surname>Marlow</surname><given-names>C</given-names> </name><name name-style="western"><surname>Adamic</surname><given-names>L</given-names> </name></person-group><article-title>The role of social networks in information diffusion</article-title><conf-name>Proceedings of the 21st International Conference on World Wide Web</conf-name><conf-date>Apr 16-20, 2012</conf-date><pub-id pub-id-type="doi">10.1145/2187836.2187907</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Eysenbach</surname><given-names>G</given-names> </name></person-group><article-title>Infodemiology: the epidemiology of (mis)information</article-title><source>Am J Med</source><year>2002</year><month>12</month><day>15</day><volume>113</volume><issue>9</issue><fpage>763</fpage><lpage>765</lpage><pub-id pub-id-type="doi">10.1016/s0002-9343(02)01473-0</pub-id><pub-id pub-id-type="medline">12517369</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Qazvinian</surname><given-names>V</given-names> </name><name name-style="western"><surname>Rosengren</surname><given-names>E</given-names> </name><name name-style="western"><surname>Radev</surname><given-names>DR</given-names> </name><name name-style="western"><surname>Barzilay</surname><given-names>MQ</given-names> </name><name name-style="western"><surname>Johnson</surname><given-names>R</given-names> </name></person-group><article-title>Rumor has it: identifying misinformation in microblogs</article-title><access-date>2025-08-04</access-date><conf-name>Proceedings of the 2011 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Jul 27-31, 2011</conf-date><conf-loc>Edinburgh, Scotland</conf-loc><fpage>1589</fpage><lpage>1599</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/D11-1147/">https://aclanthology.org/D11-1147/</ext-link></comment></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vosoughi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Roy</surname><given-names>D</given-names> </name><name name-style="western"><surname>Aral</surname><given-names>S</given-names> </name></person-group><article-title>The spread of true and false news online</article-title><source>Science</source><year>2018</year><month>03</month><day>9</day><volume>359</volume><issue>6380</issue><fpage>1146</fpage><lpage>1151</lpage><pub-id pub-id-type="doi">10.1126/science.aap9559</pub-id><pub-id pub-id-type="medline">29590045</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Depoux</surname><given-names>A</given-names> </name><name name-style="western"><surname>Martin</surname><given-names>S</given-names> </name><name name-style="western"><surname>Karafillakis</surname><given-names>E</given-names> </name><name name-style="western"><surname>Preet</surname><given-names>R</given-names> </name><name name-style="western"><surname>Wilder-Smith</surname><given-names>A</given-names> </name><name name-style="western"><surname>Larson</surname><given-names>H</given-names> </name></person-group><article-title>The pandemic of social media panic travels faster than the COVID-19 outbreak</article-title><source>J Travel Med</source><year>2020</year><month>05</month><day>18</day><volume>27</volume><issue>3</issue><fpage>taaa031</fpage><pub-id pub-id-type="doi">10.1093/jtm/taaa031</pub-id><pub-id pub-id-type="medline">32125413</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Roozenbeek</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schneider</surname><given-names>CR</given-names> </name><name name-style="western"><surname>Dryhurst</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Susceptibility to misinformation about COVID-19 around the world</article-title><source>R Soc Open Sci</source><year>2020</year><month>10</month><volume>7</volume><issue>10</issue><fpage>201199</fpage><pub-id pub-id-type="doi">10.1098/rsos.201199</pub-id><pub-id pub-id-type="medline">33204475</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Swire-Thompson</surname><given-names>B</given-names> </name><name name-style="western"><surname>Lazer</surname><given-names>D</given-names> </name></person-group><article-title>Public health and online misinformation: challenges and recommendations</article-title><source>Annu Rev Public Health</source><year>2020</year><month>04</month><day>2</day><volume>41</volume><fpage>433</fpage><lpage>451</lpage><pub-id pub-id-type="doi">10.1146/annurev-publhealth-040119-094127</pub-id><pub-id pub-id-type="medline">31874069</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Oh</surname><given-names>SH</given-names> </name><name name-style="western"><surname>Paek</surname><given-names>HJ</given-names> </name><name name-style="western"><surname>Hove</surname><given-names>T</given-names> </name></person-group><article-title>Cognitive and emotional dimensions of perceived risk characteristics, genre-specific media effects, and risk perceptions: the case of H1N1 influenza in South Korea</article-title><source>Asian J Commun</source><year>2015</year><month>01</month><day>2</day><volume>25</volume><issue>1</issue><fpage>14</fpage><lpage>32</lpage><pub-id pub-id-type="doi">10.1080/01292986.2014.989240</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McMullan</surname><given-names>RD</given-names> </name><name name-style="western"><surname>Berle</surname><given-names>D</given-names> </name><name name-style="western"><surname>Arn&#x00E1;ez</surname><given-names>S</given-names> </name><name name-style="western"><surname>Starcevic</surname><given-names>V</given-names> </name></person-group><article-title>The relationships between health anxiety, online health information seeking, and cyberchondria: systematic review and meta-analysis</article-title><source>J Affect Disord</source><year>2019</year><month>02</month><day>15</day><volume>245</volume><fpage>270</fpage><lpage>278</lpage><pub-id pub-id-type="doi">10.1016/j.jad.2018.11.037</pub-id><pub-id pub-id-type="medline">30419526</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Fox</surname><given-names>S</given-names> </name><name name-style="western"><surname>Duggan</surname><given-names>M</given-names> </name></person-group><source>Health Online 2013</source><year>2013 Jan 15</year><access-date>2026-01-26</access-date><publisher-name>Pew Research Center</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.pewresearch.org/internet/2013/01/15/health-online-2013/">https://www.pewresearch.org/internet/2013/01/15/health-online-2013/</ext-link></comment></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Diviani</surname><given-names>N</given-names> </name><name name-style="western"><surname>van den Putte</surname><given-names>B</given-names> </name><name name-style="western"><surname>Giani</surname><given-names>S</given-names> </name><name name-style="western"><surname>van Weert</surname><given-names>JC</given-names> </name></person-group><article-title>Low health literacy and evaluation of online health information: a systematic review of the literature</article-title><source>J Med Internet Res</source><year>2015</year><month>05</month><day>7</day><volume>17</volume><issue>5</issue><fpage>e112</fpage><pub-id pub-id-type="doi">10.2196/jmir.4018</pub-id><pub-id pub-id-type="medline">25953147</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chou</surname><given-names>WYS</given-names> </name><name name-style="western"><surname>Hunt</surname><given-names>YM</given-names> </name><name name-style="western"><surname>Beckjord</surname><given-names>EB</given-names> </name><name name-style="western"><surname>Moser</surname><given-names>RP</given-names> </name><name name-style="western"><surname>Hesse</surname><given-names>BW</given-names> </name></person-group><article-title>Social media use in the United States: implications for health communication</article-title><source>J Med Internet Res</source><year>2009</year><month>11</month><day>27</day><volume>11</volume><issue>4</issue><fpage>e48</fpage><pub-id pub-id-type="doi">10.2196/jmir.1249</pub-id><pub-id pub-id-type="medline">19945947</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Larson</surname><given-names>HJ</given-names> </name></person-group><article-title>The biggest pandemic risk? Viral misinformation</article-title><source>Nature</source><year>2018</year><month>10</month><volume>562</volume><issue>7727</issue><fpage>309</fpage><pub-id pub-id-type="doi">10.1038/d41586-018-07034-4</pub-id><pub-id pub-id-type="medline">30327527</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Suarez-Lledo</surname><given-names>V</given-names> </name><name name-style="western"><surname>Alvarez-Galvez</surname><given-names>J</given-names> </name></person-group><article-title>Prevalence of health misinformation on social media: systematic review</article-title><source>J Med Internet Res</source><year>2021</year><month>01</month><day>20</day><volume>23</volume><issue>1</issue><fpage>e17187</fpage><pub-id pub-id-type="doi">10.2196/17187</pub-id><pub-id pub-id-type="medline">33470931</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Islam</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Sarkar</surname><given-names>T</given-names> </name><name name-style="western"><surname>Khan</surname><given-names>SH</given-names> </name><etal/></person-group><article-title>COVID-19-related infodemic and its impact on public health: a global social media analysis</article-title><source>Am J Trop Med Hyg</source><year>2020</year><month>10</month><volume>103</volume><issue>4</issue><fpage>1621</fpage><lpage>1629</lpage><pub-id pub-id-type="doi">10.4269/ajtmh.20-0812</pub-id><pub-id pub-id-type="medline">32783794</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Du</surname><given-names>X</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>S</given-names> </name></person-group><article-title>Social media rumor refutation effectiveness: evaluation, modelling and enhancement</article-title><source>Inf Process Manag</source><year>2021</year><month>01</month><volume>58</volume><issue>1</issue><fpage>102420</fpage><pub-id pub-id-type="doi">10.1016/j.ipm.2020.102420</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pal</surname><given-names>A</given-names> </name><name name-style="western"><surname>Chua</surname><given-names>AYK</given-names> </name><name name-style="western"><surname>Hoe-Lian Goh</surname><given-names>D</given-names> </name></person-group><article-title>How do users respond to online rumor rebuttals?</article-title><source>Comput Human Behav</source><year>2020</year><month>05</month><volume>106</volume><fpage>106243</fpage><pub-id pub-id-type="doi">10.1016/j.chb.2019.106243</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Maertens</surname><given-names>R</given-names> </name><name name-style="western"><surname>Roozenbeek</surname><given-names>J</given-names> </name><name name-style="western"><surname>Basol</surname><given-names>M</given-names> </name><name name-style="western"><surname>van der Linden</surname><given-names>S</given-names> </name></person-group><article-title>Long-term effectiveness of inoculation against misinformation: three longitudinal experiments</article-title><source>J Exp Psychol Appl</source><year>2021</year><month>03</month><volume>27</volume><issue>1</issue><fpage>1</fpage><lpage>16</lpage><pub-id pub-id-type="doi">10.1037/xap0000315</pub-id><pub-id pub-id-type="medline">33017160</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>XK</given-names> </name><name name-style="western"><surname>Na</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>LKW</given-names> </name><name name-style="western"><surname>Chong</surname><given-names>M</given-names> </name><name name-style="western"><surname>Choy</surname><given-names>M</given-names> </name></person-group><article-title>Exploring how online responses change in response to debunking messages about COVID-19 on WhatsApp</article-title><source>Online Inf Rev</source><year>2022</year><month>09</month><day>26</day><volume>46</volume><issue>6</issue><fpage>1184</fpage><lpage>1204</lpage><pub-id pub-id-type="doi">10.1108/OIR-08-2021-0422</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Walter</surname><given-names>N</given-names> </name><name name-style="western"><surname>Brooks</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Saucier</surname><given-names>CJ</given-names> </name><name name-style="western"><surname>Suresh</surname><given-names>S</given-names> </name></person-group><article-title>Evaluating the impact of attempts to correct health misinformation on social media: a meta-analysis</article-title><source>Health Commun</source><year>2021</year><month>11</month><volume>36</volume><issue>13</issue><fpage>1776</fpage><lpage>1784</lpage><pub-id pub-id-type="doi">10.1080/10410236.2020.1794553</pub-id><pub-id pub-id-type="medline">32762260</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>M</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>C</given-names> </name></person-group><article-title>Hot topic recognition of health rumors based on anti-rumor articles on the WeChat official account platform: topic modeling</article-title><source>J Med Internet Res</source><year>2023</year><month>09</month><day>21</day><volume>25</volume><fpage>e45019</fpage><pub-id pub-id-type="doi">10.2196/45019</pub-id><pub-id pub-id-type="medline">37733396</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xiao</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Li</surname><given-names>X</given-names> </name></person-group><article-title>Internet rumors during the COVID-19 pandemic: dynamics of topics and public psychologies</article-title><source>Front Public Health</source><year>2021</year><volume>9</volume><fpage>788848</fpage><pub-id pub-id-type="doi">10.3389/fpubh.2021.788848</pub-id><pub-id pub-id-type="medline">34988056</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guess</surname><given-names>A</given-names> </name><name name-style="western"><surname>Nagler</surname><given-names>J</given-names> </name><name name-style="western"><surname>Tucker</surname><given-names>J</given-names> </name></person-group><article-title>Less than you think: prevalence and predictors of fake news dissemination on Facebook</article-title><source>Sci Adv</source><year>2019</year><month>01</month><volume>5</volume><issue>1</issue><fpage>eaau4586</fpage><pub-id pub-id-type="doi">10.1126/sciadv.aau4586</pub-id><pub-id pub-id-type="medline">30662946</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhao</surname><given-names>L</given-names> </name><name name-style="western"><surname>Cui</surname><given-names>H</given-names> </name><name name-style="western"><surname>Qiu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name></person-group><article-title>SIR rumor spreading model in the new media age</article-title><source>Phys A Stat Mech Appl</source><year>2013</year><month>02</month><volume>392</volume><issue>4</issue><fpage>995</fpage><lpage>1003</lpage><pub-id pub-id-type="doi">10.1016/j.physa.2012.09.030</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Castillo</surname><given-names>C</given-names> </name><name name-style="western"><surname>Mendoza</surname><given-names>M</given-names> </name><name name-style="western"><surname>Poblete</surname><given-names>B</given-names> </name></person-group><article-title>Predicting information credibility in time-sensitive social media</article-title><source>Internet Res</source><year>2013</year><month>10</month><day>14</day><volume>23</volume><issue>5</issue><fpage>560</fpage><lpage>588</lpage><pub-id pub-id-type="doi">10.1108/IntR-05-2012-0095</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cook</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lewandowsky</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ecker</surname><given-names>UKH</given-names> </name></person-group><article-title>Neutralizing misinformation through inoculation: exposing misleading argumentation techniques reduces their influence</article-title><source>PLoS One</source><year>2017</year><volume>12</volume><issue>5</issue><fpage>e0175799</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0175799</pub-id><pub-id pub-id-type="medline">28475576</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chan</surname><given-names>MPS</given-names> </name><name name-style="western"><surname>Jones</surname><given-names>CR</given-names> </name><name name-style="western"><surname>Hall Jamieson</surname><given-names>K</given-names> </name><name name-style="western"><surname>Albarrac&#x00ED;n</surname><given-names>D</given-names> </name></person-group><article-title>Debunking: a meta-analysis of the psychological efficacy of messages countering misinformation</article-title><source>Psychol Sci</source><year>2017</year><month>11</month><volume>28</volume><issue>11</issue><fpage>1531</fpage><lpage>1546</lpage><pub-id pub-id-type="doi">10.1177/0956797617714579</pub-id><pub-id pub-id-type="medline">28895452</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nyhan</surname><given-names>B</given-names> </name><name name-style="western"><surname>Reifler</surname><given-names>J</given-names> </name></person-group><article-title>When corrections fail: the persistence of political misperceptions</article-title><source>Polit Behav</source><year>2010</year><month>06</month><volume>32</volume><issue>2</issue><fpage>303</fpage><lpage>330</lpage><pub-id pub-id-type="doi">10.1007/s11109-010-9112-2</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>C</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>X</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>D</given-names> </name></person-group><article-title>Know it to defeat it: exploring health rumor characteristics and debunking efforts on chinese social media during COVID-19 crisis</article-title><conf-name>Proceedings of the 16th International AAAI Conference on Web and Social Media</conf-name><conf-date>Jun 6-9, 2022</conf-date><conf-loc>Atlanta, GA</conf-loc><fpage>1157</fpage><lpage>1168</lpage><pub-id pub-id-type="doi">10.1609/icwsm.v16i1.19366</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sicilia</surname><given-names>R</given-names> </name><name name-style="western"><surname>Lo Giudice</surname><given-names>S</given-names> </name><name name-style="western"><surname>Pei</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Pechenizkiy</surname><given-names>M</given-names> </name><name name-style="western"><surname>Soda</surname><given-names>P</given-names> </name></person-group><article-title>Twitter rumour detection in the health domain</article-title><source>Expert Syst Appl</source><year>2018</year><month>11</month><volume>110</volume><fpage>33</fpage><lpage>40</lpage><pub-id pub-id-type="doi">10.1016/j.eswa.2018.05.019</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Budak</surname><given-names>C</given-names> </name><name name-style="western"><surname>Agrawal</surname><given-names>D</given-names> </name><name name-style="western"><surname>El Abbadi</surname><given-names>A</given-names> </name></person-group><article-title>Limiting the spread of misinformation in social networks</article-title><conf-name>Proceedings of the 20th International Conference on World Wide Web</conf-name><conf-date>Mar 28 to Apr 1, 2011</conf-date><pub-id pub-id-type="doi">10.1145/1963405.1963499</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Hassan</surname><given-names>N</given-names> </name><name name-style="western"><surname>Arslan</surname><given-names>F</given-names> </name><name name-style="western"><surname>Li</surname><given-names>C</given-names> </name><name name-style="western"><surname>Tremayne</surname><given-names>M</given-names> </name></person-group><article-title>Toward automated fact-checking: detecting check-worthy factual claims by claimbuster</article-title><conf-name>Proceedings of the 23rd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</conf-name><conf-date>Aug 13-17, 2017</conf-date><pub-id pub-id-type="doi">10.1145/3097983.3098131</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>X</given-names> </name><name name-style="western"><surname>Shu</surname><given-names>C</given-names> </name></person-group><article-title>Research on the influencing factors of new media health rumor recognition ability</article-title><conf-name>2024 11th International Conference on Behavioural and Social Computing (BESC)</conf-name><conf-date>Aug 16-18, 2024</conf-date><conf-loc>Harbin, China</conf-loc><fpage>1</fpage><lpage>6</lpage><pub-id pub-id-type="doi">10.1109/BESC64747.2024.10780542</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chua</surname><given-names>AYK</given-names> </name><name name-style="western"><surname>Banerjee</surname><given-names>S</given-names> </name></person-group><article-title>To share or not to share: the role of epistemic belief in online health rumors</article-title><source>Int J Med Inform</source><year>2017</year><month>12</month><volume>108</volume><fpage>36</fpage><lpage>41</lpage><pub-id pub-id-type="doi">10.1016/j.ijmedinf.2017.08.010</pub-id><pub-id pub-id-type="medline">29132629</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pennycook</surname><given-names>G</given-names> </name><name name-style="western"><surname>Rand</surname><given-names>DG</given-names> </name></person-group><article-title>Lazy, not biased: susceptibility to partisan fake news is better explained by lack of reasoning than by motivated reasoning</article-title><source>Cognition</source><year>2019</year><month>07</month><volume>188</volume><fpage>39</fpage><lpage>50</lpage><pub-id pub-id-type="doi">10.1016/j.cognition.2018.06.011</pub-id><pub-id pub-id-type="medline">29935897</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pennycook</surname><given-names>G</given-names> </name><name name-style="western"><surname>Epstein</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Mosleh</surname><given-names>M</given-names> </name><name name-style="western"><surname>Arechar</surname><given-names>AA</given-names> </name><name name-style="western"><surname>Eckles</surname><given-names>D</given-names> </name><name name-style="western"><surname>Rand</surname><given-names>DG</given-names> </name></person-group><article-title>Shifting attention to accuracy can reduce misinformation online</article-title><source>Nature</source><year>2021</year><month>04</month><volume>592</volume><issue>7855</issue><fpage>590</fpage><lpage>595</lpage><pub-id pub-id-type="doi">10.1038/s41586-021-03344-2</pub-id><pub-id pub-id-type="medline">33731933</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="web"><article-title>Weibo announces third quarter 2024 unaudited financial results</article-title><source>Weibo</source><year>2024</year><access-date>2026-01-26</access-date><comment><ext-link ext-link-type="uri" xlink:href="http://ir.weibo.com/news-releases/news-release-details/weibo-announces-third-quarter-2024-unaudited-financial-results/">http://ir.weibo.com/news-releases/news-release-details/weibo-announces-third-quarter-2024-unaudited-financial-results/</ext-link></comment></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vraga</surname><given-names>EK</given-names> </name><name name-style="western"><surname>Bode</surname><given-names>L</given-names> </name></person-group><article-title>Using expert sources to correct health misinformation in social media</article-title><source>Sci Commun</source><year>2017</year><month>10</month><volume>39</volume><issue>5</issue><fpage>621</fpage><lpage>645</lpage><pub-id pub-id-type="doi">10.1177/1075547017731776</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>van der Meer</surname><given-names>T</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>Y</given-names> </name></person-group><article-title>Seeking formula for misinformation treatment in public health crises: the effects of corrective information type and source</article-title><source>Health Commun</source><year>2020</year><month>05</month><volume>35</volume><issue>5</issue><fpage>560</fpage><lpage>575</lpage><pub-id pub-id-type="doi">10.1080/10410236.2019.1573295</pub-id><pub-id pub-id-type="medline">30761917</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Chao</surname><given-names>F</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>G</given-names> </name></person-group><article-title>Evaluating rumor debunking effectiveness during the COVID-19 pandemic crisis: utilizing user stance in comments on Sina Weibo</article-title><source>Front Public Health</source><year>2021</year><volume>9</volume><fpage>770111</fpage><pub-id pub-id-type="doi">10.3389/fpubh.2021.770111</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Silla</surname><given-names>CN</given-names> </name><name name-style="western"><surname>Freitas</surname><given-names>AA</given-names> </name></person-group><article-title>A survey of hierarchical classification across different application domains</article-title><source>Data Min Knowl Disc</source><year>2011</year><month>01</month><volume>22</volume><issue>1-2</issue><fpage>31</fpage><lpage>72</lpage><pub-id pub-id-type="doi">10.1007/s10618-010-0175-9</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Kowsari</surname><given-names>K</given-names> </name><name name-style="western"><surname>Brown</surname><given-names>DE</given-names> </name><name name-style="western"><surname>Heidarysafa</surname><given-names>M</given-names> </name><name name-style="western"><surname>Jafari Meimandi</surname><given-names>K</given-names> </name><name name-style="western"><surname>Gerber</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Barnes</surname><given-names>LE</given-names> </name></person-group><article-title>HDLTex: hierarchical deep learning for text classification</article-title><conf-name>2017 16th IEEE International Conference on Machine Learning and Applications (ICMLA)</conf-name><conf-date>Dec 18-21, 2017</conf-date><conf-loc>Cancun, Mexico</conf-loc><fpage>364</fpage><lpage>371</lpage><pub-id pub-id-type="doi">10.1109/ICMLA.2017.0-134</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Krippendorff</surname><given-names>K</given-names> </name></person-group><article-title>Reliability in content analysis: some common misconceptions and recommendations</article-title><source>Hum Commun Res</source><year>2004</year><month>07</month><day>1</day><volume>30</volume><issue>3</issue><fpage>411</fpage><lpage>433</lpage><pub-id pub-id-type="doi">10.1111/j.1468-2958.2004.tb00738.x</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Tunstall</surname><given-names>L</given-names> </name><name name-style="western"><surname>Reimers</surname><given-names>N</given-names> </name><name name-style="western"><surname>Jo</surname><given-names>UES</given-names> </name><etal/></person-group><article-title>Efficient few-shot learning without prompts</article-title><source>arXiv</source><comment>Preprint posted online on  Sep 22, 2022</comment><pub-id pub-id-type="doi">10.48550/arXiv.2209.11055</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Grootendorst</surname><given-names>M</given-names> </name></person-group><article-title>BERTopic: neural topic modeling with a class-based TF-IDF procedure</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 11, 2022</comment><pub-id pub-id-type="doi">10.48550/arXiv.2203.05794</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Egger</surname><given-names>R</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>J</given-names> </name></person-group><article-title>A topic modeling comparison between LDA, NMF, Top2Vec, and BERTopic to demystify Twitter posts</article-title><source>Front Sociol</source><year>2022</year><volume>7</volume><fpage>886498</fpage><pub-id pub-id-type="doi">10.3389/fsoc.2022.886498</pub-id><pub-id pub-id-type="medline">35602001</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Reimers</surname><given-names>N</given-names> </name><name name-style="western"><surname>Gurevych</surname><given-names>I</given-names> </name></person-group><article-title>Sentence-BERT: sentence embeddings using siamese BERT-networks</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 27, 2019</comment><pub-id pub-id-type="doi">10.48550/arXiv.1908.10084</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="web"><source>SuperCLUE</source><access-date>2026-04-22</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.superclueai.com/homepage">https://www.superclueai.com/homepage</ext-link></comment></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>DeepSeek-AI</surname></name><name name-style="western"><surname>A</surname><given-names>Liu</given-names> </name><etal/></person-group><article-title>DeepSeek-V3.2: Pushing the Frontier of Open Large Language Models</article-title><source>arXiv</source><comment>Preprint posted online on  Dec 2, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2512.02556</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="web"><article-title>Law of the People&#x2019;s Republic of China on protection of minors</article-title><source>The National People&#x2019;s Congress of the People&#x2019;s Republic of China</source><year>2021</year><access-date>2026-01-26</access-date><comment><ext-link ext-link-type="uri" xlink:href="http://www.npc.gov.cn/englishnpc/c2759/c23934/202109/t20210914_384808.html">http://www.npc.gov.cn/englishnpc/c2759/c23934/202109/t20210914_384808.html</ext-link></comment></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="web"><person-group person-group-type="author"><collab>The Standing Committee of the National People&#x2019;s Congress</collab></person-group><article-title>Law of the People&#x2019;s Republic of China on Protection of the Rights and Interests of the Elderly (2018 amendment) [webpage in Chinese]</article-title><source>The National People&#x2019;s Congress of the People&#x2019;s Republic of China</source><year>2019</year><access-date>2026-01-28</access-date><comment><ext-link ext-link-type="uri" xlink:href="http://www.npc.gov.cn/npc/c2/c30834/201905/t20190521_296650.html">http://www.npc.gov.cn/npc/c2/c30834/201905/t20190521_296650.html</ext-link></comment></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="book"><person-group person-group-type="author"><collab>National Research Council (US) Committee on Diet and Health</collab></person-group><article-title>Extent and distribution of chronic disease: an overview</article-title><source>Diet and Health: Implications for Reducing Chronic Disease Risk</source><year>1989</year><access-date>2026-01-15</access-date><publisher-name>National Academies Press (US)</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/books/NBK218755">https://www.ncbi.nlm.nih.gov/books/NBK218755</ext-link></comment></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Breiman</surname><given-names>L</given-names> </name></person-group><article-title>Statistical modeling: the two cultures (with comments and a rejoinder by the author)</article-title><source>Statist Sci</source><year>2001</year><volume>16</volume><issue>3</issue><fpage>199</fpage><lpage>231</lpage><pub-id pub-id-type="doi">10.1214/ss/1009213726</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Guestrin</surname><given-names>C</given-names> </name></person-group><article-title>XGBoost: a scalable tree boosting system</article-title><conf-name>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</conf-name><conf-date>Aug 13-17, 2016</conf-date><conf-loc>San Francisco, CA</conf-loc><fpage>785</fpage><lpage>794</lpage><pub-id pub-id-type="doi">10.1145/2939672.2939785</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Lundberg</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>SI</given-names> </name></person-group><article-title>A unified approach to interpreting model predictions</article-title><source>arXiv</source><comment>Preprint posted online on  May 22, 2017</comment><pub-id pub-id-type="doi">10.48550/arXiv.1705.07874</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lundberg</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Erion</surname><given-names>G</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>H</given-names> </name><etal/></person-group><article-title>From local explanations to global understanding with explainable AI for trees</article-title><source>Nat Mach Intell</source><year>2020</year><volume>2</volume><issue>1</issue><fpage>56</fpage><lpage>67</lpage><pub-id pub-id-type="doi">10.1038/s42256-019-0138-9</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Romero</surname><given-names>DM</given-names> </name><name name-style="western"><surname>Meeder</surname><given-names>B</given-names> </name><name name-style="western"><surname>Kleinberg</surname><given-names>J</given-names> </name></person-group><article-title>Differences in the mechanics of information diffusion across topics: idioms, political hashtags, and complex contagion on Twitter</article-title><conf-name>WWW &#x2019;11: Proceedings of the 20th international conference on World wide web</conf-name><conf-date>Mar 28 to Apr 1, 2011</conf-date><conf-loc>New York, NY</conf-loc><fpage>695</fpage><lpage>704</lpage><pub-id pub-id-type="doi">10.1145/1963405.1963503</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Grinsztajn</surname><given-names>L</given-names> </name><name name-style="western"><surname>Oyallon</surname><given-names>E</given-names> </name><name name-style="western"><surname>Varoquaux</surname><given-names>G</given-names> </name></person-group><article-title>Why do tree-based models still outperform deep learning on typical tabular data?</article-title><conf-name>NIPS&#x2019;22: Proceedings of the 36th International Conference on Neural Information Processing Systems</conf-name><conf-date>Nov 28 to Dec 9, 2022</conf-date><conf-loc>New Orleans, LA</conf-loc><fpage>507</fpage><lpage>520</lpage><pub-id pub-id-type="doi">10.52202/068431-0037</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>R&#x00F6;der</surname><given-names>M</given-names> </name><name name-style="western"><surname>Both</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hinneburg</surname><given-names>A</given-names> </name></person-group><article-title>Exploring the space of topic coherence measures</article-title><conf-name>WSDM &#x2019;15: Proceedings of the Eighth ACM International Conference on Web Search and Data Mining</conf-name><conf-date>Feb 2-6, 2015</conf-date><pub-id pub-id-type="doi">10.1145/2684822.2685324</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dieng</surname><given-names>AB</given-names> </name><name name-style="western"><surname>Ruiz</surname><given-names>FJR</given-names> </name><name name-style="western"><surname>Blei</surname><given-names>DM</given-names> </name></person-group><article-title>Topic modeling in embedding spaces</article-title><source>Trans Assoc Comput Linguist</source><year>2020</year><month>12</month><volume>8</volume><fpage>439</fpage><lpage>453</lpage><pub-id pub-id-type="doi">10.1162/tacl_a_00325</pub-id></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Baumgartner</surname><given-names>FR</given-names> </name><name name-style="western"><surname>Jones</surname><given-names>BD</given-names> </name></person-group><source>Agendas and Instability in American Politics</source><year>2010</year><publisher-name>University of Chicago Press</publisher-name><pub-id pub-id-type="doi">10.7208/chicago/9780226039534.001.0001</pub-id><pub-id pub-id-type="other">9780226039497</pub-id></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shih</surname><given-names>TJ</given-names> </name><name name-style="western"><surname>Wijaya</surname><given-names>R</given-names> </name><name name-style="western"><surname>Brossard</surname><given-names>D</given-names> </name></person-group><article-title>Media coverage of public health epidemics: linking framing and issue attention cycle toward an integrated theory of print news coverage of epidemics</article-title><source>Mass Commun Soc</source><year>2008</year><month>04</month><day>7</day><volume>11</volume><issue>2</issue><fpage>141</fpage><lpage>160</lpage><pub-id pub-id-type="doi">10.1080/15205430701668121</pub-id></nlm-citation></ref><ref id="ref64"><label>64</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ball-Rokeach</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>DeFleur</surname><given-names>ML</given-names> </name></person-group><article-title>A dependency model of mass-media effects</article-title><source>Communic Res</source><year>1976</year><month>01</month><volume>3</volume><issue>1</issue><fpage>3</fpage><lpage>21</lpage><pub-id pub-id-type="doi">10.1177/009365027600300101</pub-id></nlm-citation></ref><ref id="ref65"><label>65</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Stone</surname><given-names>JA</given-names> </name></person-group><article-title>The sharing of pandemic-related information from U.S. government Twitter accounts</article-title><source>Am Politics Res</source><year>2024</year><month>09</month><volume>52</volume><issue>5</issue><fpage>467</fpage><lpage>483</lpage><pub-id pub-id-type="doi">10.1177/1532673X241263088</pub-id></nlm-citation></ref><ref id="ref66"><label>66</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kada</surname><given-names>A</given-names> </name><name name-style="western"><surname>Chouikh</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mellouli</surname><given-names>S</given-names> </name><name name-style="western"><surname>Prashad</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Straus</surname><given-names>SE</given-names> </name><name name-style="western"><surname>Fahim</surname><given-names>C</given-names> </name></person-group><article-title>An exploration of Canadian government officials&#x2019; COVID-19 messages and the public&#x2019;s reaction using social media data</article-title><source>PLoS One</source><year>2022</year><volume>17</volume><issue>9</issue><fpage>e0273153</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0273153</pub-id><pub-id pub-id-type="medline">36054094</pub-id></nlm-citation></ref><ref id="ref67"><label>67</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ahluwalia</surname><given-names>SC</given-names> </name><name name-style="western"><surname>Edelen</surname><given-names>MO</given-names> </name><name name-style="western"><surname>Qureshi</surname><given-names>N</given-names> </name><name name-style="western"><surname>Etchegaray</surname><given-names>JM</given-names> </name></person-group><article-title>Trust in experts, not trust in national leadership, leads to greater uptake of recommended actions during the COVID-19 pandemic</article-title><source>Risk Hazards Crisis Public Policy</source><year>2021</year><month>09</month><volume>12</volume><issue>3</issue><fpage>283</fpage><lpage>302</lpage><pub-id pub-id-type="doi">10.1002/rhc3.12219</pub-id><pub-id pub-id-type="medline">34226844</pub-id></nlm-citation></ref><ref id="ref68"><label>68</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>NK</given-names> </name><name name-style="western"><surname>Woo</surname><given-names>B</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>J</given-names> </name></person-group><article-title>Effect of international organizations&#x2019; direct engagement with the public: information source credibility and the public&#x2019;s attitudes towards COVID-19-related measures</article-title><source>Jpn J Polit Sci</source><year>2026</year><month>03</month><volume>27</volume><issue>1</issue><fpage>19</fpage><lpage>36</lpage><pub-id pub-id-type="doi">10.1017/S1468109925100170</pub-id></nlm-citation></ref><ref id="ref69"><label>69</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gollust</surname><given-names>SE</given-names> </name><name name-style="western"><surname>Nagler</surname><given-names>RH</given-names> </name><name name-style="western"><surname>Fowler</surname><given-names>EF</given-names> </name></person-group><article-title>The emergence of COVID-19 in the US: a public health and political communication crisis</article-title><source>J Health Polit Policy Law</source><year>2020</year><month>12</month><day>1</day><volume>45</volume><issue>6</issue><fpage>967</fpage><lpage>981</lpage><pub-id pub-id-type="doi">10.1215/03616878-8641506</pub-id><pub-id pub-id-type="medline">32464658</pub-id></nlm-citation></ref><ref id="ref70"><label>70</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Stifjell</surname><given-names>K</given-names> </name><name name-style="western"><surname>Sandanger</surname><given-names>TM</given-names> </name><name name-style="western"><surname>Wien</surname><given-names>C</given-names> </name></person-group><article-title>Exploring online health information-seeking behavior among young adults: scoping review</article-title><source>J Med Internet Res</source><year>2025</year><month>09</month><day>9</day><volume>27</volume><fpage>e70379</fpage><pub-id pub-id-type="doi">10.2196/70379</pub-id><pub-id pub-id-type="medline">40925001</pub-id></nlm-citation></ref><ref id="ref71"><label>71</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Maslow</surname><given-names>AH</given-names> </name></person-group><article-title>A theory of human motivation</article-title><source>Psychol Rev</source><year>1943</year><volume>50</volume><issue>4</issue><fpage>370</fpage><lpage>396</lpage><pub-id pub-id-type="doi">10.1037/h0054346</pub-id></nlm-citation></ref><ref id="ref72"><label>72</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Conrad</surname><given-names>P</given-names> </name></person-group><source>The Medicalization of Society: On the Transformation of Human Conditions into Treatable Disorders</source><year>2007</year><access-date>2026-01-26</access-date><publisher-name>Johns Hopkins University Press</publisher-name><pub-id pub-id-type="other">9780801885853</pub-id></nlm-citation></ref><ref id="ref73"><label>73</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fredrickson</surname><given-names>BL</given-names> </name><name name-style="western"><surname>Roberts</surname><given-names>TA</given-names> </name></person-group><article-title>Objectification theory: toward understanding women&#x2019;s lived experiences and mental health risks</article-title><source>Psychol Women Q</source><year>1997</year><month>06</month><volume>21</volume><issue>2</issue><fpage>173</fpage><lpage>206</lpage><pub-id pub-id-type="doi">10.1111/j.1471-6402.1997.tb00108.x</pub-id></nlm-citation></ref><ref id="ref74"><label>74</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ek</surname><given-names>S</given-names> </name></person-group><article-title>Gender differences in health information behaviour: a Finnish population-based survey</article-title><source>Health Promot Int</source><year>2015</year><month>09</month><volume>30</volume><issue>3</issue><fpage>736</fpage><lpage>745</lpage><pub-id pub-id-type="doi">10.1093/heapro/dat063</pub-id><pub-id pub-id-type="medline">23985248</pub-id></nlm-citation></ref><ref id="ref75"><label>75</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>MP</given-names> </name><name name-style="western"><surname>Viswanath</surname><given-names>K</given-names> </name><name name-style="western"><surname>Lam</surname><given-names>TH</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Chan</surname><given-names>SS</given-names> </name></person-group><article-title>Social determinants of health information seeking among Chinese adults in Hong Kong</article-title><source>PLoS One</source><year>2013</year><volume>8</volume><issue>8</issue><fpage>e73049</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0073049</pub-id><pub-id pub-id-type="medline">24009729</pub-id></nlm-citation></ref><ref id="ref76"><label>76</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zeng</surname><given-names>R</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Evans</surname><given-names>R</given-names> </name><name name-style="western"><surname>He</surname><given-names>R</given-names> </name></person-group><article-title>Pregnancy-related information seeking and sharing in the social media era among expectant mothers: qualitative study</article-title><source>J Med Internet Res</source><year>2019</year><month>12</month><day>4</day><volume>21</volume><issue>12</issue><fpage>e13694</fpage><pub-id pub-id-type="doi">10.2196/13694</pub-id><pub-id pub-id-type="medline">31799939</pub-id></nlm-citation></ref><ref id="ref77"><label>77</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Connell</surname><given-names>R</given-names> </name></person-group><source>Masculinities</source><year>2020</year><publisher-name>Routledge</publisher-name><pub-id pub-id-type="other">9781003116479</pub-id></nlm-citation></ref><ref id="ref78"><label>78</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Courtenay</surname><given-names>WH</given-names> </name></person-group><article-title>Constructions of masculinity and their influence on men&#x2019;s well-being: a theory of gender and health</article-title><source>Soc Sci Med</source><year>2000</year><month>05</month><volume>50</volume><issue>10</issue><fpage>1385</fpage><lpage>1401</lpage><pub-id pub-id-type="doi">10.1016/s0277-9536(99)00390-1</pub-id><pub-id pub-id-type="medline">10741575</pub-id></nlm-citation></ref><ref id="ref79"><label>79</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bucher</surname><given-names>T</given-names> </name></person-group><article-title>Want to be on the top? Algorithmic power and the threat of invisibility on Facebook</article-title><source>New Media &#x0026; Society</source><year>2012</year><month>11</month><volume>14</volume><issue>7</issue><fpage>1164</fpage><lpage>1180</lpage><pub-id pub-id-type="doi">10.1177/1461444812440159</pub-id></nlm-citation></ref><ref id="ref80"><label>80</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Dey</surname><given-names>M</given-names> </name></person-group><article-title>Weibo statistics by revenue, market cap, users and facts</article-title><source>Electro IQ</source><year>2025</year><access-date>2026-01-26</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://electroiq.com/stats/weibo-statistics/">https://electroiq.com/stats/weibo-statistics/</ext-link></comment></nlm-citation></ref><ref id="ref81"><label>81</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cutrona</surname><given-names>SL</given-names> </name><name name-style="western"><surname>Mazor</surname><given-names>KM</given-names> </name><name name-style="western"><surname>Vieux</surname><given-names>SN</given-names> </name><name name-style="western"><surname>Luger</surname><given-names>TM</given-names> </name><name name-style="western"><surname>Volkman</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Finney Rutten</surname><given-names>LJ</given-names> </name></person-group><article-title>Health information-seeking on behalf of others: characteristics of &#x201C;surrogate seekers&#x201D;</article-title><source>J Cancer Educ</source><year>2015</year><month>03</month><volume>30</volume><issue>1</issue><fpage>12</fpage><lpage>19</lpage><pub-id pub-id-type="doi">10.1007/s13187-014-0701-3</pub-id><pub-id pub-id-type="medline">24989816</pub-id></nlm-citation></ref><ref id="ref82"><label>82</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Petty</surname><given-names>RE</given-names> </name><name name-style="western"><surname>Cacioppo</surname><given-names>JT</given-names> </name></person-group><source>Communication and Persuasion: Central and Peripheral Routes to Attitude Change</source><year>1986</year><access-date>2026-01-26</access-date><publisher-name>Springer-Verlag</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://link.springer.com/book/10.1007/978-1-4612-4964-1">https://link.springer.com/book/10.1007/978-1-4612-4964-1</ext-link></comment><pub-id pub-id-type="other">9781461249641</pub-id></nlm-citation></ref><ref id="ref83"><label>83</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Blau</surname><given-names>P</given-names> </name></person-group><source>Exchange and Power in Social Life</source><year>2017</year><access-date>2026-01-28</access-date><publisher-name>Routledge</publisher-name><pub-id pub-id-type="other">9780887386282</pub-id></nlm-citation></ref><ref id="ref84"><label>84</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Surma</surname><given-names>J</given-names> </name></person-group><article-title>Social exchange in online social networks. The reciprocity phenomenon on Facebook</article-title><source>Comput Commun</source><year>2016</year><month>01</month><volume>73</volume><fpage>342</fpage><lpage>346</lpage><pub-id pub-id-type="doi">10.1016/j.comcom.2015.06.017</pub-id></nlm-citation></ref><ref id="ref85"><label>85</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Merton</surname><given-names>RK</given-names> </name></person-group><article-title>The Matthew effect in science. The reward and communication systems of science are considered</article-title><source>Science</source><year>1968</year><month>01</month><day>5</day><volume>159</volume><issue>3810</issue><fpage>56</fpage><lpage>63</lpage><pub-id pub-id-type="doi">10.1126/science.159.3810.56</pub-id><pub-id pub-id-type="medline">5634379</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Validation metrics for the information extraction framework (interannotator agreement, overall model performance, and category-specific performance evaluation for gender and age variables).</p><media xlink:href="jmir_v28i1e91516_app1.docx" xlink:title="DOCX File, 25 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Detailed statistics of gender and age group mentions across the top 10 BERTopic-derived themes.</p><media xlink:href="jmir_v28i1e91516_app2.docx" xlink:title="DOCX File, 26 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Top 20 characteristic keywords of health-related rumor debunking posts, stratified by gender and age.</p><media xlink:href="jmir_v28i1e91516_app3.docx" xlink:title="DOCX File, 23 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Top 20 most frequent social roles identified in health-related rumor debunking posts.</p><media xlink:href="jmir_v28i1e91516_app4.docx" xlink:title="DOCX File, 24 KB"/></supplementary-material><supplementary-material id="app5"><label>Multimedia Appendix 5</label><p>Intermodel agreement and performance metrics for sentiment classification.</p><media xlink:href="jmir_v28i1e91516_app5.docx" xlink:title="DOCX File, 24 KB"/></supplementary-material></app-group></back></article>