<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Infodemiology</journal-id><journal-id journal-id-type="publisher-id">infodemiology</journal-id><journal-id journal-id-type="index">38</journal-id><journal-title>JMIR Infodemiology</journal-title><abbrev-journal-title>JMIR Infodemiology</abbrev-journal-title><issn pub-type="epub">2564-1891</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v6i1e98319</article-id><article-id pub-id-type="doi">10.2196/98319</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Identifying Public Response Topics to the Centers for Disease Control and Prevention&#x2019;s COVID-19 Communications on Social Media: Infoveillance Study Using Large Language Model&#x2013;Based Rephrasing</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Xin</surname><given-names>Wangjiaxuan</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="pa1">&#x2020;</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yin</surname><given-names>Shuhua</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Park</surname><given-names>Albert</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ge</surname><given-names>Yaorong</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Chen</surname><given-names>Shi</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff4">4</xref></contrib></contrib-group><aff id="aff1"><institution>College of Computing and Informatics, The University of North Carolina at Charlotte</institution><addr-line>9201 University City Blvd</addr-line><addr-line>Charlotte</addr-line><addr-line>NC</addr-line><country>United States</country></aff><aff id="aff2"><institution>Department of Public Health and Health Administration, The University of North Carolina at Charlotte</institution><addr-line>Charlotte</addr-line><addr-line>NC</addr-line><country>United States</country></aff><aff id="aff3"><institution>School of Health Information Science, University of Victoria</institution><addr-line>Victoria</addr-line><addr-line>BC</addr-line><country>Canada</country></aff><aff id="aff4"><institution>School of Data Science, The University of North Carolina at Charlotte</institution><addr-line>Charlotte</addr-line><addr-line>NC</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Mackey</surname><given-names>Tim</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Lau</surname><given-names>Gabriel Rongyang</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Son</surname><given-names>Hyunsang</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Gevorgyan</surname><given-names>Natalya</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Wangjiaxuan Xin, MSc, College of Computing and Informatics, The University of North Carolina at Charlotte, 9201 University City Blvd, Charlotte, NC, 28223, United States; <email>wxin@charlotte.edu</email></corresp><fn fn-type="other" id="pa1"><label>&#x2020;</label><p>College of Computing and Informatics, The University of North Carolina at Charlotte, Charlotte, North Carolina, United States</p></fn><fn fn-type="other" id="pa2"><label>&#x2021;</label><p>College of Computing and Informatics, The University of North Carolina at Charlotte, Charlotte, North Carolina, United States</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>8</day><month>10</month><year>2026</year></pub-date><volume>6</volume><elocation-id>e98319</elocation-id><history><date date-type="received"><day>15</day><month>04</month><year>2026</year></date><date date-type="rev-recd"><day>25</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>31</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Wangjiaxuan Xin, Shuhua Yin, Albert Park, Yaorong Ge, Shi Chen. Originally published in JMIR Infodemiology (<ext-link ext-link-type="uri" xlink:href="https://infodemiology.jmir.org">https://infodemiology.jmir.org</ext-link>), 8.10.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Infodemiology, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://infodemiology.jmir.org/">https://infodemiology.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://infodemiology.jmir.org/2026/1/e98319"/><abstract><sec><title>Background</title><p>Public health agencies increasingly use social media platforms such as Twitter (subsequently rebranded X, X Corp) to monitor public responses to official health communications and support timely decision-making during health crises. Public replies to communications from official agencies provide valuable insights into population-level discourse and engagement with health policies. However, analyzing large-scale short-text responses remains challenging because social media replies are often brief, informal, ambiguous, and linguistically fragmented. Topic modeling offers a scalable approach for identifying thematic patterns in such data, but its performance is often limited when applied to social media short texts. Recent advances in large language models (LLMs) provide new opportunities to improve text normalization before topic modeling, but their value and limitations for public health infoveillance remain underexplored.<inline-graphic xlink:href="infodemiology_v6i1e98319_fig03.png"/></p></sec><sec><title>Objective</title><p>This study develops and evaluates TM-Rephrase, a model-agnostic LLM-based rephrasing framework designed to improve the performance of topic models for short public health social media texts while ensuring rephrasing preserves the original semantic meaning, stance or intent, and tone.</p></sec><sec sec-type="methods"><title>Methods</title><p>We analyzed 25,027 public replies to official Centers for Disease Control and Prevention tweets on Twitter collected between May 2020 and November 2022. TM-Rephrase transformed informal and noisy short texts into more standardized expressions using 2 prompt-guided schemes: general rephrasing and colloquial-to-formal rephrasing. Original and rephrased texts were analyzed using multiple topic models, with rephrasing generated by Gemini 2.5 Flash, GPT&#x2011;4o mini, and Mistral-7B-Instruct. Topic quality was evaluated using coherence, uniqueness, redundancy, and diversity. We also conducted an expert semantic-fidelity validation study in which 4 expert raters evaluated whether rephrased texts preserved the original meaning, stance or intent, and tone.</p></sec><sec sec-type="results"><title>Results</title><p>TM-Rephrase generally improved topic quality across models, rephrasing schemes, and LLM backbones. For latent Dirichlet allocation, <inline-formula><mml:math id="ieqn1"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> coherence increased from 0.3094 without rephrasing to 0.5004 with colloquial-to-formal rephrasing. For BERTopic (BERT-based topic modeling), <inline-formula><mml:math id="ieqn2"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> coherence improved from 0.4078 to 0.4734. Diversity-related metrics also improved in most rephrasing settings. Additional checks using <inline-formula><mml:math id="ieqn3"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mi>M</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, <inline-formula><mml:math id="ieqn4"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>U</mml:mi><mml:mi>C</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, and <inline-formula><mml:math id="ieqn5"><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>S</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> generally supported the main improvement pattern. Sensitivity analysis across different topic numbers further suggested that the coherence gains were not limited to the main setting. Expert validation showed favorable semantic preservation overall, with a mean rating of 3.90 (SD 1.01) and median of 4.00 (IQR 3.00-5.00) of 5.00, and acceptable interrater reliability, intraclass correlation coefficient =0.736. General rephrasing received stronger semantic-fidelity ratings than colloquial-to-formal rephrasing, suggesting that stronger formalization may improve topic readability while introducing greater risk of changes to the original meaning, stance or intent, and tone.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>TM-Rephrase can enhance the coherence, distinctiveness, and interpretability of topic modeling outputs for short, noisy public health social media data. However, it is a text-normalization preprocessing strategy for topic enhancement rather than a fully neutral replacement for original public discourse.</p></sec></abstract><kwd-group><kwd>text mining</kwd><kwd>topic modeling</kwd><kwd>public health infoveillance</kwd><kwd>COVID-19</kwd><kwd>social media analytics</kwd><kwd>large language models</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction<inline-graphic xlink:href="infodemiology_v6i1e98319_fig02.png"/></title><sec id="s1-1"><title>Background</title><p>The wide adoption of social media has fundamentally transformed public discourse, particularly during public health crises (eg, the COVID-19 pandemic) [<xref ref-type="bibr" rid="ref1">1</xref>]. Twitter, characterized by rapid information dissemination and a concise conversational format, has emerged as a critical yet methodologically challenging data source for understanding and analyzing public discussion topics and the dynamics of information exchange during the COVID-19 pandemic [<xref ref-type="bibr" rid="ref2">2</xref>]. In addition to enabling real-time communication, these platforms support large-scale public engagement, where individuals actively express opinions, share experiences, and respond to official health communications. For public health agencies such as the US Centers for Disease Control and Prevention (CDC), analyzing short-text public replies that express opinions and concerns provides valuable insights for informing more effective communication strategies to the public [<xref ref-type="bibr" rid="ref3">3</xref>]. Such innovative surveillance and analyses can help capture evolving public concerns, assess responses to current health policies, and improve the alignment between institutional messaging and public needs. However, from a public health perspective, effectively leveraging these data remains challenging because of the unstructured nature of short, user-generated social media texts, which complicates the systematic monitoring and timely interpretation of population-level responses. In particular, the inherent brevity, informality, and ambiguity of individual tweets, along with the 280-character limit [<xref ref-type="bibr" rid="ref4">4</xref>], pose significant challenges for traditional social media data mining and analytics. These characteristics are particularly problematic for topic modeling, which aims to identify latent thematic structures in textual corpora, because limited textual context constrains reliable semantic interpretation.</p><p>The contextual sparsity of social media short texts often leads to topics of keywords that are less coherent and less diverse (ie, more overlapping topics with redundant or unrelated keywords), thereby reducing interpretability and limiting their utility for understanding various public opinions during the pandemic. This limitation is further compounded by the heterogeneous nature of online public discourse, where variations in writing and expression styles introduce additional complexity into textual data. This problem becomes particularly pronounced when attempting to capture emerging and latent thematic structures in communications during public health emergencies, where semantic ambiguity, brevity, and informal linguistic styles (eg, abbreviations, slang, layman expressions, and misspellings) are prevalent [<xref ref-type="bibr" rid="ref5">5</xref>]. Moreover, even when topic models identify clusters of topic keywords, they often lack meaningful interpretability without additional contextualization or external knowledge from experts, making it difficult to translate model outputs into coherent narratives for public health decision support. Consequently, these challenges limit the effectiveness of topic modeling in supporting systematic analysis of public health discourse. These issues suggest the need to standardize and formalize short social media texts into more contextually formal content before the application of topic modeling in health communications, thereby improving semantic clarity and enhancing the interpretability of downstream analytical results.</p><p>Topic modeling, as a widely used method for discovering latent thematic structures in textual corpora, is exemplified by the classic probabilistic model, latent Dirichlet allocation (LDA), proposed by Blei et al [<xref ref-type="bibr" rid="ref6">6</xref>] in 2003. However, despite its broad adoption, applying LDA or its variants to short texts from social media platforms poses substantial challenges. The brevity, informality, and noise of tweets weaken statistical signals and exacerbate the context sparsity, leading to incoherent, redundant, or less interpretable topics generated from these probabilistic models [<xref ref-type="bibr" rid="ref7">7</xref>].</p><p>To address these challenges, some enrichment strategies have been proposed. One line of research augments short texts with external knowledge or auxiliary contextual signals to enhance statistical robustness [<xref ref-type="bibr" rid="ref8">8</xref>]. Another direction leverages distributed representations of words and documents, incorporating embeddings or neural architectures to map texts into richer semantic representations. Prominent examples include BERTopic (BERT-based topic modeling) [<xref ref-type="bibr" rid="ref9">9</xref>] and Top2Vec [<xref ref-type="bibr" rid="ref10">10</xref>], which use more recent transformer-based embeddings and clustering methods to generate more semantically meaningful topics. Recent work has explored more advanced neural network architectures, such as FASTopic (fast, adaptive, stable, and transferable topic model) [<xref ref-type="bibr" rid="ref11">11</xref>] and topic-semantic contrastive learning [<xref ref-type="bibr" rid="ref12">12</xref>], which demonstrate improved handling of context sparsity and have the potential to be applied in health communications.</p><p>However, all these approaches focus on algorithmic model redesign with parameterized inference strategies for short texts, which may adapt poorly to real-life large-scale data (eg, public health-related discourse on social media). In addition, such topic modeling algorithms typically output sets of top-ranked keywords that can be difficult for analysts or readers without additional domain knowledge to interpret.</p><p>Regarding the public discourse during COVID-19, the pandemic fundamentally transformed social media into the primary platform for real-time public discourse, where Twitter has been extensively used to study public attitudes and evolving concerns toward the pandemic and various health policies [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref14">14</xref>]. Scholars have analyzed public reactions to official communications from health agencies such as the CDC, focusing on opinion and thematic feedback to health policies [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. Additional research has explored geospatial and demographic variations in public concerns, demonstrating the heterogeneous impact of the pandemic on different communities [<xref ref-type="bibr" rid="ref16">16</xref>]. Despite the active research and the value of social media for effective public health infoveillance during emergencies, it often overlooks the interpretability issues arising from short and noisy texts in online discourse, especially on social media platforms.</p><p>Motivated by Maini et al [<xref ref-type="bibr" rid="ref17">17</xref>], we leverage large language models (LLMs) as a text preprocessing approach for rephrasing short texts for topic modeling, aiming to standardize and formalize public replies to CDC tweets on Twitter. This framework, termed TM-Rephrase, builds on recent advances in LLMs, which have demonstrated powerful performance across diverse text-processing tasks, including text summarization [<xref ref-type="bibr" rid="ref18">18</xref>], sentiment analysis [<xref ref-type="bibr" rid="ref19">19</xref>], and information retrieval [<xref ref-type="bibr" rid="ref20">20</xref>].</p><p>TM-Rephrase differs from conventional text preprocessing approaches. Standard text normalization and spelling correction primarily address surface-level noise, such as misspellings, inconsistent casing, punctuation, or informal abbreviations. Data augmentation usually creates additional training examples to improve model learning. In contrast, TM-Rephrase uses LLM-based, prompt-constrained rephrasing to transform each short social media reply into a more standardized and contextually explicit expression for topic modeling. Its distinguishing characteristic lies in treating rephrasing as a model-agnostic, data-centric preprocessing strategy for improving topic-model outputs rather than as a replacement for topic modeling algorithms or as generic text rewriting.</p><p>In this study, we examine the impact of TM-Rephrase on the quality of topics derived from 4 representative topic models, using a dataset of public replies to CDC Twitter tweets during the COVID-19 pandemic. We hypothesize that rephrasing original short-text public replies on social media through TM-Rephrase enhances topic quality in terms of both intratopic semantic coherence and inter&#x2013;topic metrics of topic model outputs, which are represented as sets of keywords. Specifically, we investigate two rephrasing schemes: (1) general rephrasing and (2) colloquial-to-formal rephrasing, and evaluate their effectiveness across multiple topic modeling algorithms, including probabilistic algorithms (eg, LDA [<xref ref-type="bibr" rid="ref6">6</xref>]) and more advanced neural network-based topic models (eg, BERTopic [<xref ref-type="bibr" rid="ref9">9</xref>] and FASTopic [<xref ref-type="bibr" rid="ref11">11</xref>]).</p></sec><sec id="s1-2"><title>Research Questions and Objectives</title><p>The research questions (RQs) of this study are as follows: (RQ1) whether and how effectively TM-Rephrase improves the quality and interpretability of topic model outputs for short texts in health communications, (RQ2) how different rephrasing schemes influence topic model performance, and (RQ3) how to interpret public discussions (ie, replies to CDC during the COVID-19 pandemic) based on the topics and inform public health practice.</p><p>This study aims to improve the analysis of public responses to official public health communications on social media to support more effective public health surveillance. To achieve this, we develop and evaluate TM-Rephrase, an LLM-based model-agnostic framework that enhances the quality and interpretability of topic modeling for short-text public replies to CDC communications during the COVID-19 pandemic.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>TM-Rephrase is designed as a multistage pipeline to improve topic modeling performance in public health communications, as shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>. The pipeline is composed of four major stages: (1) data collection, (2) LLM-based rephrasing, (3) data preprocessing, and (4) topic modeling with a selected algorithm and evaluation of the resulting topics, as detailed in the following subsections. We used this pipeline to study how rephrased text of public responses to CDC tweets on Twitter impacts topic modeling outcomes compared to nonrephrased ones.<inline-graphic xlink:href="infodemiology_v6i1e98319_fig01.png"/></p><p>Before introducing the model details, we begin by formulating the problem setting with preliminary notations that are used throughout this paper. Let <inline-formula><mml:math id="ieqn6"><mml:mi>D</mml:mi><mml:mo>=</mml:mo><mml:mo>{</mml:mo><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi> </mml:mi><mml:mi> </mml:mi><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi> </mml:mi><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mi> </mml:mi><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msub><mml:mo>}</mml:mo></mml:math></inline-formula> denote a corpus of <italic>N</italic> short-text tweets (ie, replies to CDC tweets). Each document <inline-formula><mml:math id="ieqn7"><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is typically brief and context-limited because of the inherent sparsity of social media discourse.</p><p>The goal of topic modeling in this setting is to extract and model a set of <italic>K</italic> latent topics, denoted by <inline-formula><mml:math id="ieqn8"><mml:mi>T</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi> </mml:mi><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mi> </mml:mi><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula>, where each topic <inline-formula><mml:math id="ieqn9"><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is represented by a set of its top-<italic>T</italic> representative keywords, extracted or generated by the corresponding topic modeling algorithm, that is, <inline-formula><mml:math id="ieqn10"><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi> </mml:mi><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mi> </mml:mi><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi> </mml:mi><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mi>K</mml:mi><mml:mo>.</mml:mo></mml:math></inline-formula> These representative keywords serve as the semantic expression of the topic and form the basis for downstream topic interpretation, evaluation, and further analysis.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Overview of the TM-Rephrase framework. The pipeline integrates data collection, LLM&#x2013;based rephrasing, data preprocessing, and topic modeling algorithm application with evaluation.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="infodemiology_v6i1e98319_fig04.png"/></fig></sec><sec id="s2-2"><title>Data Collection</title><p>We constructed a dataset of public responses to official CDC communications on Twitter during the COVID-19 pandemic, covering the period from May 2020 to November 2022. Data were collected via the Twitter Academic API and search queries (Section 1: data search queries in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), tracking 7 official CDC accounts: @CDCgov, @CDCDirector, @CDCGlobal, @CDCMMWR, @CDCtravel, @DrReddCDC, and @CDCEmergency. The initial retrieval yielded 180,090 public interactions associated with these CDC accounts, including replies and direct mentions. The retrieved metadata included tweet posting dates, tweet text, tweet account ID, account type, public engagement metrics, referenced tweet type, and referenced tweet IDs. Amazon Web Services was used as the computing or storage environment for managing the collected data. For the purpose of this study, only tweets that could be linked to an original CDC tweet were retained. Approximately 49,000 direct mentions that referenced a CDC account but were not replies to a specific CDC tweet were removed. For example, a tweet that mentions @CDCEmergency independently from a user&#x2019;s own account was not considered a reply to the CDC communication. Another 104,000 reply-like interactions were further excluded as they could not be reliably linked to an accessible original CDC tweet, primarily because the original CDC tweet was deleted, unavailable, or otherwise inaccessible at the time of data processing. This step was necessary to preserve conversational context between each public reply and the corresponding CDC communication.</p><p>The final analytic dataset consisted of 25,027 public replies linked to 1,512 unique original CDC tweets. These replies capture public opinions, concerns, and discussions in response to official CDC communications during a major public health emergency. We retained only English-language tweets. Statistics of the final original and rephrased datasets are presented in <xref ref-type="table" rid="table1">Table 1</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Statistics of the original and rephrased datasets of public replies to CDC<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> communications during the COVID-19 pandemic.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Quantity of original and rephrased datasets</td><td align="left" valign="bottom">Original</td><td align="left" valign="bottom">General rephrased</td><td align="left" valign="bottom">C-to-f<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> rephrased</td></tr></thead><tbody><tr><td align="left" valign="top">Mean number of words</td><td align="left" valign="top">27.33</td><td align="left" valign="top">27.76</td><td align="left" valign="top">28.14</td></tr><tr><td align="left" valign="top">SD of word counts</td><td align="left" valign="top">14.23</td><td align="left" valign="top">15.12</td><td align="left" valign="top">14.11</td></tr><tr><td align="left" valign="top">Minimum number of words</td><td align="left" valign="top">1</td><td align="left" valign="top">1</td><td align="left" valign="top">1</td></tr><tr><td align="left" valign="top">25th percentile number of words</td><td align="left" valign="top">15</td><td align="left" valign="top">16</td><td align="left" valign="top">16</td></tr><tr><td align="left" valign="top">Number of words, median (IQR)</td><td align="left" valign="top">27 (15-40)</td><td align="left" valign="top">27 (16-41)</td><td align="left" valign="top">29 (16-42)</td></tr><tr><td align="left" valign="top">75th percentile number of words</td><td align="left" valign="top">40</td><td align="left" valign="top">41</td><td align="left" valign="top">42</td></tr><tr><td align="left" valign="top">Maximum number of words</td><td align="left" valign="top">64</td><td align="left" valign="top">66</td><td align="left" valign="top">68</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>CDC: US Centers for Disease Control and Prevention.</p></fn><fn id="table1fn2"><p><sup>b</sup>C-to-f: colloquial to formal.</p></fn></table-wrap-foot></table-wrap><p>Formally, the dataset <inline-formula><mml:math id="ieqn11"><mml:mi>D</mml:mi><mml:mo>=</mml:mo><mml:mo>{</mml:mo><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi> </mml:mi><mml:mi> </mml:mi><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi> </mml:mi><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mi> </mml:mi><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msub><mml:mo>}</mml:mo></mml:math></inline-formula> denotes the original corpus, where each short-text reply is denoted as:</p><disp-formula id="E1"><label>(1)</label><mml:math id="eqn1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>d</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msubsup><mml:mi>w</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msubsup><mml:mi>w</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msubsup><mml:mi>w</mml:mi><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>N</mml:mi></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>which contains a limited number of tokens <inline-formula><mml:math id="ieqn12"><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> of tweet <inline-formula><mml:math id="ieqn13"><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> as each tweet reply is limited to 280 characters [<xref ref-type="bibr" rid="ref4">4</xref>]. The short and informal nature of these short texts results in sparse lexical co-occurrence signals, motivating the need for semantic enrichment before topic modeling algorithms are applied.</p></sec><sec id="s2-3"><title>LLM-Based Rephrasing</title><p>The second stage introduces an LLM-based rephrasing operator that transforms each short-text document (ie, a public reply to a CDC tweet) into a semantically and contextually refined version that is similar to CDC&#x2019;s official communication. Let <inline-formula><mml:math id="ieqn14"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denote a rephrase model (ie, an LLM) parameterized by <italic>&#x03B8;</italic>. For each short-text reply <inline-formula><mml:math id="ieqn15"><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, the rephrasing process is defined as:</p><disp-formula id="E2"><label>(2)</label><mml:math id="eqn2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mrow><mml:mover><mml:mi>d</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>&#x03C0;</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mi>N</mml:mi></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <inline-formula><mml:math id="ieqn16"><mml:mover accent="true"><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:math></inline-formula> refers to the rephrased short-text reply, and &#x03C0; denotes a carefully designed prompt that constrains the generation process to preserve semantic fidelity and ensure that rephrasing is executed under a certain scheme that will be introduced in this subsection. This rephrasing stage is crucial for the downstream topic modeling of the public replies and for better understanding of public responses to official health communications during emergencies.</p><p>For the rephrasing stage, we experimented with 3 different LLMs: Google Gemini (Gemini 2.5 Flash [<xref ref-type="bibr" rid="ref21">21</xref>]), OpenAI generative pretrained transformer (GPT&#x2011;4o mini [<xref ref-type="bibr" rid="ref22">22</xref>]), and an open-source LLM (Mistral-7B-Instruct [<xref ref-type="bibr" rid="ref23">23</xref>]), given their strong natural language generation capabilities and complementary trade-offs between performance and cost for large-scale applications. We used a low sampling temperature (0.2) to ensure semantic fidelity and minimize stochastic variation during automated rephrasing, consistent with findings from prior work [<xref ref-type="bibr" rid="ref24">24</xref>].</p><p>To systematically investigate how different styles of linguistic refinement influence topic modeling performance, TM-Rephrase incorporates 2 distinct prompt-guided rephrasing schemes that we developed: general rephrasing and colloquial-to-formal rephrasing. Although both schemes are designed to preserve semantic fidelity, they differ in the degree and style of linguistic transformation applied to the original short texts of public replies. Prompts for these 2 schemes are shown in Table S1 of <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> in Section 2: prompt information for rephrasing schemes.</p><p>The general rephrasing scheme performs lightweight linguistic refinement while strictly maintaining the meaning and informational content of the original text. Its primary objective is to improve grammatical correctness, syntactic clarity, and overall readability without altering domain-specific terminology, named entities, hashtags, usernames, or technical details. Concretely, the general rephrasing scheme fixes typographical errors, incomplete sentence fragments, inconsistent verb tenses, and ambiguous phrasing, all of which are common in social media discourse. It may restructure fragmented clauses into complete sentences and resolve minor ambiguities that arise from informal writing conventions. However, it avoids introducing new concepts, removing key terms, or substituting specialized vocabulary. As such, general rephrasing can be viewed as a minimal-intervention normalization process that enhances structural regularity while preserving the lexical distribution of domain-relevant tokens.</p><p>The colloquial-to-formal rephrasing scheme transforms informal, conversational, or slang-based expressions into formal, professional English suitable for public health communication contexts. The rephrased outputs are structured to resemble the tone and style of public health reports or professional summaries, using complete sentences and standardized grammar while avoiding slang, contractions, and casual phrasing. During this process, implicit references&#x2014;such as omitted subjects, abbreviated expressions, or context-dependent meanings (eg, &#x201C;this shot,&#x201D; &#x201C;they said,&#x201D; or fragmentary statements lacking clear referents)&#x2014;are made explicit by clarifying the intended subject, action, or relationship within the sentence. Fragmented or elliptical constructions are rewritten into fully articulated and logically coherent statements to improve readability and interpretability. Strict constraints are applied to preserve the original meaning, including retaining all named entities, hashtags, usernames, and domain-specific terminology. No additional claims, interpretations, or external information are introduced.</p><p>These two rephrasing schemes enable a controlled comparison between moderate linguistic normalization and more substantial stylistic formalization for health communications. This design allows us to evaluate whether incremental grammatical refinement alone suffices to improve topic modeling performance, or whether deeper formalization is required for optimal topic quality. In order to ensure semantic alignment between the original and rephrased texts, the prompt explicitly instructs the LLMs to retain meaning and avoid modifying domain-specific content.</p><p>The resulting rephrased corpus is defined in <xref ref-type="disp-formula" rid="E3">equation 3</xref>, where <inline-formula><mml:math id="ieqn17"><mml:mover accent="true"><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:math></inline-formula> represents each individual rephrased short-text reply:</p><disp-formula id="E3"><label>(3)</label><mml:math id="eqn3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mover><mml:mi>D</mml:mi><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mo>=</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mi>d</mml:mi><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mi>d</mml:mi><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mi>d</mml:mi><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mi>N</mml:mi></mml:msub><mml:mo fence="false" stretchy="false">}</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula></sec><sec id="s2-4"><title>Expert Validation of Semantic Fidelity</title><p>To evaluate whether LLM-based rephrasing preserved the meaning, stance or intent, and tone of the original public replies, we further conducted a human expert validation study. A sample of 30 original CDC-reply tweets was randomly selected for validation. For each original tweet, 2 rephrased versions were evaluated: one generated using the general rephrasing scheme and one generated using the colloquial-to-formal scheme, resulting in 60 original-rephrased text pairs across 2 schemes.</p><p>Four expert raters with complementary expertise in health communication, social media research, linguistics, and public health independently evaluated the rephrased texts. Each rater compared the original tweet with the corresponding rephrased version in a paired format and provided a single holistic semantic-fidelity rating using a 5-point Likert scale. Raters were instructed to consider whether the rephrased version preserved the original meaning, stance or intent, and tone, but these dimensions were not rated separately. As the task required direct comparison between original and rephrased texts, complete blinding to the rephrasing scheme was not feasible. Raters completed the evaluations independently. We summarized the validation results using the mean, median, and distribution of expert ratings. To assess interrater reliability, we calculated a 2-way random-effects, absolute-agreement intraclass correlation coefficient (ICC[2,4]) for the aggregated expert ratings.</p></sec><sec id="s2-5"><title>Data Preprocessing</title><p>Before applying topic modeling, both the original and rephrased corpora undergo a consistent and standardized preprocessing procedure to ensure comparability and reduce extraneous noise. The preprocessing step includes the removal of URLs, punctuation marks, and stop words, which do not contribute meaningful semantic information to topic inference during topic modeling.</p><p>At the same time, semantically meaningful elements are carefully preserved. Emojis are retained through their textual descriptions to maintain their affective or contextual signals. For example, the emoji &#x1F60A; is converted to a textual representation &#x201C;smiling face,&#x201D; and &#x1F637; is mapped to &#x201C;face with medical mask.&#x201D; This conversion ensures that nontextual, yet informative elements remain encoded within the lexical space rather than be discarded as noise.</p><p>All texts are further normalized by converting all characters to lowercase, followed by tokenization and lemmatization. Lowercasing eliminates case-based lexical duplication, tokenization segments text into analyzable units, and lemmatization reduces inflected forms of tokens to their base form, thereby consolidating semantically equivalent variants into unified lexical representations. All these operations promote lexical consistency and stabilize downstream word co-occurrence statistics in topic modeling.</p><p>Formally, let <inline-formula><mml:math id="ieqn18"><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x03B3;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mo>&#x2219;</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> denote the preprocessing operator parameterized by &#x03B3;, which encapsulates the sequence of normalization, tokenization, and lemmatization in the data preprocessing step. After preprocessing, each original document <inline-formula><mml:math id="ieqn19"><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and rephrased document <inline-formula><mml:math id="ieqn20"><mml:mover accent="true"><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:math></inline-formula> are respectively converted into the preprocessed format:</p><disp-formula id="E4"><label>(4)</label><mml:math id="eqn4"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>p</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mi>y</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mrow><mml:mover><mml:mi>p</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mi>y</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mi>d</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mi>N</mml:mi></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>Here <inline-formula><mml:math id="ieqn21"><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mi>V</mml:mi><mml:mo>|</mml:mo></mml:mrow></mml:msup></mml:math></inline-formula> and <inline-formula><mml:math id="ieqn22"><mml:mover accent="true"><mml:mrow><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mi>V</mml:mi><mml:mo>|</mml:mo></mml:mrow></mml:msup></mml:math></inline-formula> denote the tokenized representations of the original and the rephrased documents (ie, short-text replies), respectively, under a shared vocabulary space <italic>V</italic>. Maintaining identical preprocessing across both corpora ensures that any observed differences in topic modeling performance can be credited to the rephrasing rather than discrepancies in preprocessing or representation learning.</p></sec><sec id="s2-6"><title>Topic Models</title><p>In the final stage of the pipeline, topic modeling algorithms are applied, and their performance is systematically evaluated. As illustrated in <xref ref-type="fig" rid="figure2">Figure 2</xref>, both the original and rephrased replies are modeled by the same topic models, and the resulting topics are compared and evaluated through quantitative metrics and expert semantic fidelity evaluation. The original corpus <inline-formula><mml:math id="ieqn23"><mml:mi>D</mml:mi></mml:math></inline-formula> and rephrased corpus <inline-formula><mml:math id="ieqn24"><mml:mover accent="true"><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:math></inline-formula> are independently input into the same specific topic modeling algorithm <inline-formula><mml:math id="ieqn25"><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> to ensure methodological consistency:</p><disp-formula id="E5"><label>(5)</label><mml:math id="eqn5"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>T</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>T</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>D</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mrow><mml:mover><mml:mi>T</mml:mi><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>T</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mover><mml:mi>D</mml:mi><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>This parallel modeling design yields 2 corresponding sets of topic outputs, that is, one derived from the original texts and the other from the rephrased texts. This enables a direct and controlled comparison. Due to the same topic modeling procedure, hyperparameter settings for topic models and preprocessing configurations are maintained across both corpora. Therefore, any observed differences in topic quality can be credited to the rephrasing scheme rather than the specific topic modeling algorithm.</p><p>To comprehensively evaluate the effects of TM-Rephrase, 4 representative topic models were applied, including conventional statistical and neural network-based topic models. This suite of different topic models enables a systematic analysis of how different topic modeling algorithms respond to TM-Rephrase.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Outline of the topic evaluation process. Both the original and rephrased tweets were modeled using the same topic models, and the resulting topics were compared and evaluated through quantitative metrics and expert semantic fidelity evaluation.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="infodemiology_v6i1e98319_fig05.png"/></fig><p>LDA [<xref ref-type="bibr" rid="ref6">6</xref>] is a foundational probabilistic topic model serving as the benchmark in topic modeling studies. It is effective for classic topic discovery but is usually challenged by data sparsity in short social media texts.</p><p>BERTopic [<xref ref-type="bibr" rid="ref9">9</xref>] is a neural topic model using contextual embeddings (eg, Sentence-BERT [<xref ref-type="bibr" rid="ref25">25</xref>]) and density-based clustering. It is considered well-suited for noisy, informal social media data.</p><p>FASTopic [<xref ref-type="bibr" rid="ref11">11</xref>] is based on a transformer backbone that directly captures semantic relationships between document embeddings and learnable topic-word embeddings. It uses an embedding transport plan as an optimization objective to enhance topic-word and document-topic associations.</p><p>Topic-semantic contrastive topic model (TSCTM) [<xref ref-type="bibr" rid="ref12">12</xref>] addresses short-text sparsity via contrastive learning, generating dense semantic vectorized representations and topic distributions.</p><p>Together, these different topic models ensure that the evaluation of TM-Rephrase is not limited to a single topic modeling paradigm, but captures methodological robustness across multiple topic models. For all 4 topic models, we fixed the number of topics (<italic>K</italic>) at 8, based on an exploratory analysis to balance thematic detail and interpretability for the study dataset. For each topic, represented as a set of keywords, we retained the top 15 keywords , ranked according to the model&#x2019;s topic-word probability. Details of the experimental implementation are available in Section S3: implementation details in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-7"><title>Topic Modeling Performance Evaluation</title><sec id="s2-7-1"><title>Overview</title><p>Four quantitative evaluation metrics were computed to quantify topic coherence, uniqueness, redundancy, and diversity, with details shown below.</p></sec><sec id="s2-7-2"><title><italic>Topic Coherence</italic> (<inline-formula><mml:math id="ieqn26"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mi> </mml:mi></mml:math></inline-formula>)</title><p><inline-formula><mml:math id="ieqn27"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> [<xref ref-type="bibr" rid="ref26">26</xref>] measures the semantic coherence among the topic keywords, which are determined by topic-word probabilities. This metric was chosen based on the mathematical analysis and justifications in the study by Wu [<xref ref-type="bibr" rid="ref27">27</xref>]. This metric fundamentally relies on normalized pointwise mutual information (NPMI) to quantify pairwise word semantic associations. For any 2 words <inline-formula><mml:math id="ieqn28"><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="ieqn29"><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, their NPMI score is defined as:</p><disp-formula id="E6"><label>(6)</label><mml:math id="eqn6"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mi>M</mml:mi><mml:mi>I</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mi>&#x03F5;</mml:mi></mml:mrow><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mi>&#x03F5;</mml:mi></mml:mrow></mml:mfrac><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mi>&#x03F5;</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>Here, <inline-formula><mml:math id="ieqn30"><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="ieqn31"><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> are the probabilities of observing words <inline-formula><mml:math id="ieqn32"><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="ieqn33"><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, respectively, and <inline-formula><mml:math id="ieqn34"><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi> </mml:mi><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> is the co-occurrence probability of the word pair <inline-formula><mml:math id="ieqn35"><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi> </mml:mi><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> within a defined external corpus (the Wikipedia [Wikimedia Foundation, Inc] corpus was used in this study [<xref ref-type="bibr" rid="ref28">28</xref>]). A small constant <inline-formula><mml:math id="ieqn36"><mml:mi>&#x03F5;</mml:mi></mml:math></inline-formula> (eg, <inline-formula><mml:math id="ieqn37"><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>12</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>) is added to avoid division by 0 or taking the logarithm of 0. Building on the NPMI formulation, the <inline-formula><mml:math id="ieqn38"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> coherence score can be formally expressed as follows:</p><disp-formula id="E7"><label>(7)</label><mml:math id="eqn7"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>T</mml:mi></mml:mfrac><mml:munderover><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:munderover><mml:mi>cos</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="bold">v</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mi>M</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mrow><mml:mi mathvariant="bold">v</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mi>M</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:msubsup><mml:mrow><mml:mo>{</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>}</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>Here, <inline-formula><mml:math id="ieqn39"><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mi>M</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> represents the vector of NPMI scores between a word <inline-formula><mml:math id="ieqn40"><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and all other words in the topic keyword list, defined as:</p><disp-formula id="E8"><label>(8)</label><mml:math id="eqn8"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>v</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mi>M</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mi>M</mml:mi><mml:mi>I</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>Similarly, the aggregated NPMI vector across all words in the topic keyword list is defined as:</p><disp-formula id="E9"><label>(9)</label><mml:math id="eqn9"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi mathvariant="bold-italic">v</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mi>M</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:msubsup><mml:mrow><mml:mo>{</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>}</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:munderover><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:munderover><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mi>M</mml:mi><mml:mi>I</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:mstyle></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <italic>T</italic> is the number of top keywords in the topic (<italic>T</italic>=15 in this study). The cosine similarity is then computed between the word-specific and aggregated vectors, and the final <inline-formula><mml:math id="ieqn41"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> score is obtained by averaging across all words. A higher <inline-formula><mml:math id="ieqn42"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> score indicates stronger semantic coherence and better interpretability of the topics.</p></sec><sec id="s2-7-3"><title>Topic Uniqueness</title><p>Nan et al [<xref ref-type="bibr" rid="ref29">29</xref>] proposed topic uniqueness (<italic>TU</italic>), a metric that quantifies how distinct the top keywords within each topic are across the entire topic set <inline-formula><mml:math id="ieqn43"><mml:mi>T</mml:mi></mml:math></inline-formula>. This helps identify whether individual topics use distinctive vocabulary. Given <italic>K</italic> topics and the top <italic>T</italic> keywords of each topic, <italic>TU</italic> is computed as:</p><disp-formula id="E10"><label>(10)</label><mml:math id="eqn10"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>T</mml:mi><mml:mi>U</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>K</mml:mi></mml:mfrac><mml:munderover><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>T</mml:mi></mml:mfrac><mml:munder><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munder><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mi mathvariant="normal">#</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <inline-formula><mml:math id="ieqn44"><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denotes the top keyword set of the <italic>k</italic>-th topic, and <inline-formula><mml:math id="ieqn45"><mml:mo>#</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> denotes the occurrence count of word <inline-formula><mml:math id="ieqn46"><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> in the top <italic>T</italic> words of all topics. <italic>TU</italic> ranges from <italic>1</italic> /K to 1, and a higher <italic>TU</italic> score indicates more unique topics.</p></sec><sec id="s2-7-4"><title>Topic Redundancy</title><p>Topic redundancy (<italic>TR</italic>) [<xref ref-type="bibr" rid="ref30">30</xref>] was developed to measure the overlap of top keywords across different topics. The formula for <italic>TR</italic> is given as:</p><disp-formula id="E11"><label>(11)</label><mml:math id="eqn11"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>T</mml:mi><mml:mi>R</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>K</mml:mi></mml:mfrac><mml:munderover><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>T</mml:mi></mml:mfrac><mml:munder><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munder><mml:mfrac><mml:mrow><mml:mi mathvariant="normal">#</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>A lower <italic>TR</italic> score suggests that the topics are more distinct.</p></sec><sec id="s2-7-5"><title>Topic Diversity</title><p>The formula for the topic diversity <italic>(TD)</italic> [<xref ref-type="bibr" rid="ref31">31</xref>], which measures the proportion of unique top keywords across all topics, is defined as:</p><disp-formula id="E12"><label>(12)</label><mml:math id="eqn12"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>T</mml:mi><mml:mi>D</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>K</mml:mi></mml:mfrac><mml:munderover><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:munderover><mml:mfrac><mml:mn>1</mml:mn><mml:mi>T</mml:mi></mml:mfrac><mml:munder><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munder><mml:mi>I</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi mathvariant="normal">#</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>A higher <italic>TD</italic> score indicates more diverse topics with less word overlap. This metric works by identifying words that are unique to a single topic&#x2019;s top keyword list. The indicator function <inline-formula><mml:math id="ieqn47"><mml:mi>I</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mo>&#x2219;</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> is defined by a stepwise function. This function returns 1 if and only if a word <inline-formula><mml:math id="ieqn48"><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> appears in the top keyword list of exactly 1 topic, and 0 otherwise.</p><p>These 4 metrics, <inline-formula><mml:math id="ieqn49"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, <italic>TU</italic>, <italic>TR,</italic> and <italic>TD</italic>, evaluate topic quality across complementary dimensions. In brief, <inline-formula><mml:math id="ieqn50"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> measures semantic coherence among a topic&#x2019;s top keywords, <italic>TU</italic> assesses topic uniqueness across topics, <italic>TR</italic> quantifies topic redundancy due to topical overlap, and <italic>TD</italic> evaluates whether the model produces a diverse set of topics. These metrics provide a comprehensive set of measures for evaluating and comparing topics generated from original nonrephrased tweets and those rephrased through 2 different TM-Rephrase schemes (general and colloquial-to-formal rephrasing).</p><p>Through systematic comparisons between generated topics based on original and rephrased texts respectively, we evaluated the extent to which the model-agnostic TM-Rephrase framework improves the intratopic coherence and intertopic quality.</p></sec></sec><sec id="s2-8"><title>Ethical Considerations</title><p>This study analyzed publicly accessible online tweets from Twitter that were replies to official CDC communications during the COVID-19 pandemic. To comply with the platform&#x2019;s privacy policy, data collection was limited to publicly accessible tweets. The social media data component did not involve direct interaction with Twitter users, recruitment of social media users, intervention, or collection of private information beyond publicly available tweet content. Informed consent from social media users was not obtained because the analysis was based solely on publicly accessible online tweets.</p><p>This study, including the human expert validation component, was reviewed by the Office of Research Protections and Integrity at the University of North Carolina at Charlotte and received a Notice of Determination of Exemption. This study was determined to meet the exempt category cited under 45 CFR 46.104(d), Exemption Category 2. The institutional review board study number is IRB-26&#x2010;1225, titled &#x201C;Human Evaluation of Topic Modeling Performance for Text Analysis of Public Health-Related Social Media Short Texts,&#x201D; with an approval date of July 8, 2026 (Figure S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> in Section S4: notice of determination of exemption from IRB Office in UNC Charlotte).</p><p>To minimize privacy risks, no usernames, profile information, or other direct user identifiers are reported in this paper. Example tweets, when presented, are used only for methodological illustration, and the main findings are reported in aggregate form. Expert validation results are also summarized in aggregate, without identifying individual raters.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Overview</title><p>This section presents the results of quantitative evaluation of TM-Rephrase, followed by illustrative examples. We report results across four topic models, FASTopic, BERTopic, TSCTM, and LDA, under three conditions, as follows: (1) original tweets without any rephrasing, (2) general rephrasing, and (3) colloquial-to-formal rephrasing, with three different LLMs used for rephrasing (ie, Gemini 2.5 Flash, GPT&#x2011;4o mini, and Mistral-7B-Instruct). Topic quality was quantitatively assessed using 4 metrics: <inline-formula><mml:math id="ieqn51"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, <italic>TU</italic>, <italic>TR</italic>, and <italic>TD,</italic> as described in the Topic Modeling Performance Evaluation section.<inline-graphic xlink:href="infodemiology_v6i1e98319_fig06.png"/></p><p>Before introducing the quantitative results, we refer to <xref ref-type="table" rid="table1">Table 1</xref> to assess potential text expansion or compression caused by TM-Rephrase. Both general and colloquial-to-formal rephrasing schemes hardly change token lengths: the mean increases marginally from 27.33 (original, SD 14.23) to 27.76 (general, SD 15.12) and 28.14 (colloquial-to-formal, SD 14.11), and the median changes slightly from 27 (IQR 15-40) to 29 (colloquial-to-formal, IQR 16-42). Quartiles and maxima show similar minor differences, indicating that rephrasing preserves the overall text length distribution, and any effects on topic modeling are unlikely to stem from token counts.</p></sec><sec id="s3-2"><title>Quantitative Results</title><sec id="s3-2-1"><title>Overview</title><p><xref ref-type="table" rid="table2">Table 2</xref> reports the quantitative evaluation results under a 3-factor design, including 4 topic modeling algorithms, 3 LLM backends, and 2 rephrasing schemes plus the original baseline, resulting in a total of 28 configurations. This design enables a systematic assessment of the main effects of rephrasing, as well as its interaction with model architecture and LLM choice.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Quantitative metric results for various topic models, both with and without TM-Rephrase by three LLMs<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup><sup>,</sup><sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom"/><td align="left" valign="bottom"><inline-formula><mml:math id="ieqn52"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>(0&#x2010;1) &#x2191;<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td><td align="left" valign="bottom"><italic>TU<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></italic> (0.125&#x2010;1) &#x2191;</td><td align="left" valign="bottom"><italic>TR<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup></italic> (0&#x2010;1) &#x2193;</td><td align="left" valign="bottom"><italic>TD<sup><xref ref-type="table-fn" rid="table2fn6">f</xref></sup></italic> (0&#x2010;1) &#x2191;</td></tr></thead><tbody><tr><td align="left" valign="top">FASTopic<sup><xref ref-type="table-fn" rid="table2fn7">g</xref></sup> w/o<sup><xref ref-type="table-fn" rid="table2fn8">h</xref></sup> rephr<sup><xref ref-type="table-fn" rid="table2fn9">i</xref></sup><sup>,</sup><sup><xref ref-type="table-fn" rid="table2fn10">j</xref></sup></td><td align="left" valign="top">.3388</td><td align="left" valign="top">.9917</td><td align="left" valign="top">.0024</td><td align="left" valign="top">.975</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini 2.5 Flash</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FASTopic w/<sup><xref ref-type="table-fn" rid="table2fn11">k</xref></sup> general rephr<sup><xref ref-type="table-fn" rid="table2fn10">j</xref></sup></td><td align="left" valign="top">.3723</td><td align="left" valign="top">.9917</td><td align="left" valign="top">.0024</td><td align="left" valign="top">.9917</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FASTopic w/ c-to-f rephr<sup><xref ref-type="table-fn" rid="table2fn10">j</xref></sup></td><td align="left" valign="top">.3301</td><td align="left" valign="top">1<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">0<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">1<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT&#x2011;4o mini</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FASTopic w/ general rephr</td><td align="left" valign="top">.3746<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">.9917</td><td align="left" valign="top">.0024</td><td align="left" valign="top">.9917</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FASTopic w/ c-to-f rephr</td><td align="left" valign="top">.3422</td><td align="left" valign="top">1<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">0<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">1<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral-7B-Instruct</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FASTopic w/ general rephr</td><td align="left" valign="top">.3728</td><td align="left" valign="top">.9917</td><td align="left" valign="top">.0024</td><td align="left" valign="top">.9917</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FASTopic w/ c-to-f rephr</td><td align="left" valign="top">.3321</td><td align="left" valign="top">.9917</td><td align="left" valign="top">.0024</td><td align="left" valign="top">.9917</td></tr><tr><td align="left" valign="top">BERTopic<sup><xref ref-type="table-fn" rid="table2fn13">m</xref></sup> w/o rephr</td><td align="left" valign="top">.4078</td><td align="left" valign="top">.4667</td><td align="left" valign="top">.3976</td><td align="left" valign="top">.4667</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini 2.5 Flash</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>BERTopic w/ general rephr</td><td align="left" valign="top">.4564</td><td align="left" valign="top">.525<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">.3357<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">.525<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>BERTopic w/ c-to-f rephr</td><td align="left" valign="top">.4734<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">.525<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">.3452</td><td align="left" valign="top">.525<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT&#x2011;4o mini</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>BERTopic w/ general rephr</td><td align="left" valign="top">.4612</td><td align="left" valign="top">.4667</td><td align="left" valign="top">.3357<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">.4667</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>BERTopic w/ c-to-f rephr</td><td align="left" valign="top">.4687</td><td align="left" valign="top">.525<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">.3452</td><td align="left" valign="top">.525<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral-7B-Instruct</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>BERTopic w/ general rephr</td><td align="left" valign="top">.4574</td><td align="left" valign="top">.4667</td><td align="left" valign="top">.3357<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">.4667</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>BERTopic w/ c-to-f rephr</td><td align="left" valign="top">.4713</td><td align="left" valign="top">.4667</td><td align="left" valign="top">.3976</td><td align="left" valign="top">.4667</td></tr><tr><td align="left" valign="top">TSCTM<sup><xref ref-type="table-fn" rid="table2fn14">n</xref></sup> w/o rephr</td><td align="left" valign="top">.3094</td><td align="left" valign="top">.9917</td><td align="left" valign="top">.0024</td><td align="left" valign="top">.975</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini 2.5 Flash</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>TSCTM w/ general rephr</td><td align="left" valign="top">.3145</td><td align="left" valign="top">.9833</td><td align="left" valign="top">.0048</td><td align="left" valign="top">.9833</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>TSCTM w/ c-to-f rephr</td><td align="left" valign="top">.3394<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">1<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">0<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">1<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT&#x2011;4o mini</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>TSCTM w/ general rephr</td><td align="left" valign="top">.3267</td><td align="left" valign="top">.9917</td><td align="left" valign="top">.0024</td><td align="left" valign="top">.9917</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>TSCTM w/ c-to-f rephr</td><td align="left" valign="top">.3331</td><td align="left" valign="top">1<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">0<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">1<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral-7B-Instruct</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>TSCTM w/ general rephr</td><td align="left" valign="top">.3104</td><td align="left" valign="top">.9833</td><td align="left" valign="top">.0024</td><td align="left" valign="top">.9833</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>TSCTM w/ c-to-f rephr</td><td align="left" valign="top">.3213</td><td align="left" valign="top">.9917</td><td align="left" valign="top">.0024</td><td align="left" valign="top">.975</td></tr><tr><td align="left" valign="top">LDA<sup><xref ref-type="table-fn" rid="table2fn15">o</xref></sup> w/o rephr</td><td align="left" valign="top">.3094</td><td align="left" valign="top">.575<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">.3095</td><td align="left" valign="top">.575<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini 2.5 Flash</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LDA w/ general rephr</td><td align="left" valign="top">.4206</td><td align="left" valign="top">.5583</td><td align="left" valign="top">.3214</td><td align="left" valign="top">.5583</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LDA w/ c-to-f rephr</td><td align="left" valign="top">.5004<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">.575<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">.3048<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">.575<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT&#x2011;4o mini</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LDA w/ general rephr</td><td align="left" valign="top">.4197</td><td align="left" valign="top">.5583</td><td align="left" valign="top">.3214</td><td align="left" valign="top">.5583</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LDA w/ c-to-f rephr</td><td align="left" valign="top">.4917</td><td align="left" valign="top">.575<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">.3095</td><td align="left" valign="top">.575<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral-7B-Instruct</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LDA w/ general rephr</td><td align="left" valign="top">.4008</td><td align="left" valign="top">.5583</td><td align="left" valign="top">.3095</td><td align="left" valign="top">.5583</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LDA w/ c-to-f rephr</td><td align="left" valign="top">.4762</td><td align="left" valign="top">.5583</td><td align="left" valign="top">.3095</td><td align="left" valign="top">.5583</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>LLM: large language model.</p></fn><fn id="table2fn2"><p><sup>b</sup>Table rows of TM-Rephrase are results based on Gemini 2.5 Flash, GPT&#x2011;4o mini, and Mistral-7B-Instruct, respectively.</p></fn><fn id="table2fn3"><p><sup>c</sup>Arrows in the header indicate the desired direction for each metric (higher &#x2191; or lower &#x2193; is better).</p></fn><fn id="table2fn4"><p><sup>d</sup>TU: topic uniqueness.</p></fn><fn id="table2fn5"><p><sup>e</sup>TR: topic redundancy.</p></fn><fn id="table2fn6"><p><sup>f</sup>TD: topic diversity.</p></fn><fn id="table2fn7"><p><sup>g</sup>FASTopic: fast, adaptive, stable, and transferable topic model.</p></fn><fn id="table2fn8"><p><sup>h</sup>w/o: without.</p></fn><fn id="table2fn9"><p><sup>i</sup>Rephr stands for the rephrasing.</p></fn><fn id="table2fn10"><p><sup>j</sup>General rephr stands for the general rephrasing scheme, while c-to-f rephr stands for the colloquial-to-formal rephrasing scheme.</p></fn><fn id="table2fn11"><p><sup>k</sup>w: with.</p></fn><fn id="table2fn12"><p><sup>l</sup>The best results within each model group.</p></fn><fn id="table2fn13"><p><sup>m</sup>BERTopic: BERT-based topic modeling.</p></fn><fn id="table2fn14"><p><sup>n</sup>TSCTM: topic-semantic contrastive topic model.</p></fn><fn id="table2fn15"><p><sup>o</sup>LDA: latent Dirichlet allocation.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2-2"><title><italic>Effect of Rephrasing Schemes</italic></title><p>Across most configurations, TM-Rephrase improves topic quality, with the most pronounced and stable gains observed in topic coherence (<inline-formula><mml:math id="ieqn53"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>). However, the magnitude and consistency of improvement vary across topic models, evaluation metrics, and rephrasing schemes. In particular, the colloquial-to-formal rephrasing scheme generally yields the highest coherence scores. For example, with LDA, coherence increases substantially from 0.3094 (without rephrasing) to 0.5004 under Gemini 2.5 Flash, with similarly large improvements observed for GPT&#x2011;4o mini (0.4917) and Mistral-7B-Instruct (0.4762). These results indicate that rephrasing effectively mitigates contextual sparsity in short texts and enhances semantic consistency across topics.</p><p>As <inline-formula><mml:math id="ieqn54"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> coherence relies on word co-occurrence statistics derived from an external reference corpus, its values may be influenced by how closely the topic keywords align with the lexical patterns of that corpus. In this study, the reference corpus was Wikipedia, which may better represent standardized or encyclopedic expressions than informal social media language. Therefore, colloquial-to-formal rephrasing may partly improve <inline-formula><mml:math id="ieqn55"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> by shifting topic keywords toward lexical patterns more characteristic of Wikipedia, rather than solely by improving the substantive representation of public meaning.</p></sec><sec id="s3-2-3"><title><italic>Effect Across Topic Models</italic></title><p>The magnitude of improvement varies across topic models. Probabilistic models such as LDA exhibit the largest gains, reflecting their strong dependence on lexical co-occurrence patterns. BERTopic shows consistent but more moderate improvements, with coherence increasing across all LLMs and rephrasing schemes, and reaching its highest value under Gemini 2.5 Flash (<inline-formula><mml:math id="ieqn56"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>=0.4734). In contrast, FASTopic demonstrates relatively smaller and less consistent gains, with slight decreases in coherence under the colloquial-to-formal scheme in some cases (eg, <inline-formula><mml:math id="ieqn57"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>=0.3301 under Gemini 2.5 Flash). This suggests that embedding-based models may already partially capture semantic relationships despite informal language, reducing the effectiveness of input standardization. TSCTM, meanwhile, benefits substantially from rephrasing in terms of diversity-related metrics, achieving perfect topic separation (<italic>TU</italic>=1, <italic>TD</italic>=1, and <italic>TR</italic>=0) under colloquial-to-formal rephrasing with both Gemini 2.5 Flash and GPT&#x2011;4o mini.</p></sec><sec id="s3-2-4"><title><italic>Effect of LLM Backbones</italic></title><p>The improvements from TM-Rephrase are broadly similar across all 3 LLM backbones. Although minor variations in absolute performance are observed, the patterns remain clear and consistent: rephrasing improves coherence and diversity metrics across topic models regardless of the underlying LLM. For example, LDA coherence gains and TSCTM diversity improvements are observed consistently across Gemini 2.5 Flash, GPT&#x2011;4o mini, and Mistral-7B-Instruct. This stability suggests that the effectiveness of TM-Rephrase is not dependent on a specific LLM, but rather reflects a generalizable rephrasing strategy.</p></sec><sec id="s3-2-5"><title><italic>Diversity and Redundancy Metrics</italic></title><p>In addition to improving intratopic coherence, rephrasing also improves intertopic quality. Across FASTopic, BERTopic, and TSCTM, <italic>TU</italic> and <italic>TD</italic> generally increase, while <italic>TR</italic> decreases or remains similar. The most notable improvement is observed in TSCTM under colloquial-to-formal rephrasing, where perfect diversity scores are achieved, indicating fully distinct topic clusters. These findings suggest that rephrasing enhances not only intratopic coherence but also intertopic separation.</p><p><xref ref-type="fig" rid="figure3">Figure 3</xref> shows that colloquial-to-formal rephrasing often achieves larger coherence gains than general rephrasing for LDA and BERTopic, while the overall improvement patterns are observed across Gemini 2.5 Flash, GPT&#x2011;4o mini, and Mistral-7B-Instruct. These results demonstrate the robustness of the TM-Rephrase framework as well.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p><inline-formula><mml:math id="ieqn58"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> improvement (%) across large language models (LLMs) and rephrasing schemes for LDA and BERTopic. LDA: latent Dirichlet allocation.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="infodemiology_v6i1e98319_fig07.png"/></fig><p>Additionally, a nuanced trade-off between intratopic coherence and inter-<italic>TD</italic> emerges, especially for the colloquial-to-formal rephrasing scheme. While this scheme often delivers the highest coherence and perfect diversity scores, it can occasionally reduce <italic>TU</italic> in certain topic models. For example, colloquial-to-formal rephrasing maximizes coherence across all LLMs but slightly lowers <italic>TU</italic> compared to the nonrephrased baseline in LDA. This suggests that aggressive formalization might homogenize lexical choices across topics, highlighting the importance of matching rephrasing strength with model characteristics. In addition, the comparison across LLM backbones reveals a high degree of robustness in TM-Rephrase. While Gemini 2.5 Flash often achieves the strongest absolute results, GPT&#x2011;4o mini and Mistral-7B-Instruct closely track its performance across all metrics and models. The relative ranking of rephrasing schemes and topic models remains largely unchanged, indicating that TM-Rephrase is not tightly tied to a specific LLM but instead constitutes a generalizable topic-model-agnostic paradigm.</p><p>To demonstrate whether the observed improvements were dependent on the use of <inline-formula><mml:math id="ieqn59"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> coherence alone, we further evaluated topic coherence using 3 additional metrics: <inline-formula><mml:math id="ieqn60"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mi>M</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> [<xref ref-type="bibr" rid="ref26">26</xref>], <inline-formula><mml:math id="ieqn61"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>U</mml:mi><mml:mi>C</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> [<xref ref-type="bibr" rid="ref32">32</xref>], and <inline-formula><mml:math id="ieqn62"><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>S</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> [<xref ref-type="bibr" rid="ref33">33</xref>]. These metrics provide complementary views of topic coherence based on word co-occurrence patterns, with higher values indicating better coherence for all 3 metrics. The results are reported in <xref ref-type="table" rid="table3">Table 3</xref>.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Quantitative coherence metric results for various topic models, both with and without TM-Rephrase by 3 LLMs<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup><sup>,</sup><sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom"/><td align="left" valign="bottom"><inline-formula><mml:math id="ieqn63"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mi>M</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> (-1, 1) &#x2191;<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="bottom"><inline-formula><mml:math id="ieqn64"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>U</mml:mi><mml:mi>C</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> (-&#x221E;, +&#x221E;) &#x2191;</td><td align="left" valign="bottom"><inline-formula><mml:math id="ieqn65"><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>S</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> (-&#x221E;, 0] &#x2191;</td></tr></thead><tbody><tr><td align="left" valign="bottom">FASTopic<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup> w/o<sup><xref ref-type="table-fn" rid="table3fn5">e</xref></sup> rephr<sup><xref ref-type="table-fn" rid="table3fn6">f</xref></sup></td><td align="left" valign="bottom">&#x2212;0.033</td><td align="left" valign="bottom">&#x2212;1.477</td><td align="left" valign="bottom">&#x2212;5.035</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini 2.5 Flash</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FASTopic w/<sup><xref ref-type="table-fn" rid="table3fn7">g</xref></sup> general rephr<sup><xref ref-type="table-fn" rid="table3fn8">h</xref></sup></td><td align="left" valign="top">&#x2212;0.021<sup><xref ref-type="table-fn" rid="table3fn9">i</xref></sup></td><td align="left" valign="top">&#x2212;1.461</td><td align="left" valign="top">&#x2212;4.671<sup><xref ref-type="table-fn" rid="table3fn9">i</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FASTopic w/ c-to-f rephr<sup><xref ref-type="table-fn" rid="table3fn8">h</xref></sup></td><td align="left" valign="top">&#x2212;0.036</td><td align="left" valign="top">&#x2212;1.456<sup><xref ref-type="table-fn" rid="table3fn9">i</xref></sup></td><td align="left" valign="top">&#x2212;4.818</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT&#x2011;4o mini</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FASTopic w/ general rephr</td><td align="left" valign="top">&#x2212;0.027</td><td align="left" valign="top">&#x2212;1.471</td><td align="left" valign="top">&#x2212;4.833</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FASTopic w/ c-to-f rephr</td><td align="left" valign="top">&#x2212;0.031</td><td align="left" valign="top">&#x2212;1.466</td><td align="left" valign="top">&#x2212;4.874</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral-7B-Instruct</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FASTopic w/ general rephr</td><td align="left" valign="top">&#x2212;0.024</td><td align="left" valign="top">&#x2212;1.469</td><td align="left" valign="top">&#x2212;4.726</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>FASTopic w/ c-to-f rephr</td><td align="left" valign="top">&#x2212;0.035</td><td align="left" valign="top">&#x2212;1.465</td><td align="left" valign="top">&#x2212;4.819</td></tr><tr><td align="left" valign="top">BERTopic<sup><xref ref-type="table-fn" rid="table3fn10">j</xref></sup> w/o rephr</td><td align="left" valign="top">.028</td><td align="left" valign="top">.38</td><td align="left" valign="top">&#x2212;2.358</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini 2.5 Flash</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>BERTopic w/ general rephr</td><td align="left" valign="top">.049<sup><xref ref-type="table-fn" rid="table3fn9">i</xref></sup></td><td align="left" valign="top">.401</td><td align="left" valign="top">&#x2212;2.18</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>BERTopic w/ c-to-f rephr</td><td align="left" valign="top">.045</td><td align="left" valign="top">.386</td><td align="left" valign="top">&#x2212;2.163<sup><xref ref-type="table-fn" rid="table3fn9">i</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT&#x2011;4o mini</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>BERTopic w/ general rephr</td><td align="left" valign="top">.031</td><td align="left" valign="top">.411<sup><xref ref-type="table-fn" rid="table3fn9">i</xref></sup></td><td align="left" valign="top">&#x2212;2.212</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>BERTopic w/ c-to-f rephr</td><td align="left" valign="top">.042</td><td align="left" valign="top">.387</td><td align="left" valign="top">&#x2212;2.319</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral-7B-Instruct</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>BERTopic w/ general rephr</td><td align="left" valign="top">.044</td><td align="left" valign="top">.385</td><td align="left" valign="top">&#x2212;2.338</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>BERTopic w/ c-to-f rephr</td><td align="left" valign="top">.03</td><td align="left" valign="top">.381</td><td align="left" valign="top">&#x2212;2.314</td></tr><tr><td align="left" valign="top">TSCTM<sup><xref ref-type="table-fn" rid="table3fn11">k</xref></sup> w/o rephr</td><td align="left" valign="top">&#x2212;0.021</td><td align="left" valign="top">&#x2212;1.065</td><td align="left" valign="top">&#x2212;3.512</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini 2.5 Flash</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>TSCTM w/ general rephr</td><td align="left" valign="top">&#x2212;0.017</td><td align="left" valign="top">&#x2212;0.955</td><td align="left" valign="top">&#x2212;3.141</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>TSCTM w/ c-to-f rephr</td><td align="left" valign="top">&#x2212;0.013<sup><xref ref-type="table-fn" rid="table3fn9">i</xref></sup></td><td align="left" valign="top">&#x2212;0.876<sup><xref ref-type="table-fn" rid="table3fn9">i</xref></sup></td><td align="left" valign="top">&#x2212;2.949</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT&#x2011;4o mini</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>TSCTM w/ general rephr</td><td align="left" valign="top">&#x2212;0.015</td><td align="left" valign="top">&#x2212;1.032</td><td align="left" valign="top">&#x2212;2.941<sup><xref ref-type="table-fn" rid="table3fn9">i</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>TSCTM w/ c-to-f rephr</td><td align="left" valign="top">&#x2212;0.016</td><td align="left" valign="top">&#x2212;0.944</td><td align="left" valign="top">&#x2212;2.997</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral-7B-Instruct</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>TSCTM w/ general rephr</td><td align="left" valign="top">&#x2212;0.02</td><td align="left" valign="top">&#x2212;0.912</td><td align="left" valign="top">&#x2212;3.217</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>TSCTM w/ c-to-f rephr</td><td align="left" valign="top">&#x2212;0.018</td><td align="left" valign="top">&#x2212;0.998</td><td align="left" valign="top">&#x2212;3.111</td></tr><tr><td align="left" valign="top">LDA<sup><xref ref-type="table-fn" rid="table3fn12">l</xref></sup> w/o rephr</td><td align="left" valign="top">.017</td><td align="left" valign="top">&#x2212;0.084</td><td align="left" valign="top">&#x2212;2.457</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gemini 2.5 Flash</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LDA w/ general rephr</td><td align="left" valign="top">.038</td><td align="left" valign="top">.282</td><td align="left" valign="top">&#x2212;2.401</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LDA w/ c-to-f rephr</td><td align="left" valign="top">.071<sup><xref ref-type="table-fn" rid="table3fn9">i</xref></sup></td><td align="left" valign="top">.848<sup><xref ref-type="table-fn" rid="table3fn9">i</xref></sup></td><td align="left" valign="top">&#x2212;2.146</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>GPT&#x2011;4o mini</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LDA w/ general rephr</td><td align="left" valign="top">.069</td><td align="left" valign="top">.193</td><td align="left" valign="top">&#x2212;2.39</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LDA w/ c-to-f rephr</td><td align="left" valign="top">.058</td><td align="left" valign="top">.764</td><td align="left" valign="top">&#x2212;2.418</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mistral-7B-Instruct</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LDA w/ general rephr</td><td align="left" valign="top">.047</td><td align="left" valign="top">.242</td><td align="left" valign="top">&#x2212;2.06<sup><xref ref-type="table-fn" rid="table3fn9">i</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>LDA w/ c-to-f rephr</td><td align="left" valign="top">.066</td><td align="left" valign="top">.656</td><td align="left" valign="top">&#x2212;2.279</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>LLM: large language model.</p></fn><fn id="table3fn2"><p><sup>b</sup>Table rows of TM-Rephrase are results based on Gemini 2.5 Flash, GPT&#x2011;4o mini, and Mistral-7B-Instruct, respectively.</p></fn><fn id="table3fn3"><p><sup>c</sup>Arrows in the header indicate the desired direction for each metric (higher &#x2191; or lower &#x2193; is better).</p></fn><fn id="table3fn4"><p><sup>d</sup>FASTopic: fast, adaptive, stable, and transferable topic model.</p></fn><fn id="table3fn5"><p><sup>e</sup>w/o: without.</p></fn><fn id="table3fn6"><p><sup>f</sup>Rephr stands for rephrasing.</p></fn><fn id="table3fn7"><p><sup>g</sup>w/: with.</p></fn><fn id="table3fn8"><p><sup>h</sup>General rephr stands for the general rephrasing scheme, while c-to-f rephr stands for the colloquial-to-formal rephrasing scheme.</p></fn><fn id="table3fn9"><p><sup>i</sup>The best results within each model group.</p></fn><fn id="table3fn10"><p><sup>j</sup>BERTopic: BERT-based topic modeling.</p></fn><fn id="table3fn11"><p><sup>k</sup>TSCTM: topic-semantic contrastive topic model.</p></fn><fn id="table3fn12"><p><sup>l</sup>LDA: latent Dirichlet allocation.</p></fn></table-wrap-foot></table-wrap><p>The results support the main finding that TM-Rephrase generally improves topic coherence, although the magnitude of improvement varies across topic models, LLM backbones, and rephrasing schemes. The greatest and most consistent improvements are observed for LDA. For example, under Gemini 2.5 Flash, LDA improves from 0.017 to 0.071 on <inline-formula><mml:math id="ieqn66"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mi>M</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, from &#x2212;0.084 to 0.848 on <inline-formula><mml:math id="ieqn67"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>U</mml:mi><mml:mi>C</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, and from &#x2212;2.457 to &#x2212;2.146 on <inline-formula><mml:math id="ieqn68"><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>S</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> after colloquial-to-formal rephrasing. Similar improvements are also observed for GPT&#x2011;4o mini and Mistral-7B-Instruct, suggesting that the coherence gains for LDA are not specific to a single LLM backbone.</p><p>For BERTopic and TSCTM, the results also show improved coherence after rephrasing, although the best-performing rephrasing scheme differs by metric and LLM. For BERTopic, rephrased texts improve <inline-formula><mml:math id="ieqn69"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mi>M</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="ieqn70"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>U</mml:mi><mml:mi>C</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> in most settings, with Gemini 2.5 Flash general rephrasing achieving the highest <inline-formula><mml:math id="ieqn71"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mi>M</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> score and GPT&#x2011;4o mini general rephrasing achieving the highest <inline-formula><mml:math id="ieqn72"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>U</mml:mi><mml:mi>C</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> score. For TSCTM, both general and colloquial-to-formal rephrasing improve coherence relative to the baseline without rephrasing across most metrics, with colloquial-to-formal rephrasing under Gemini 2.5 Flash performing best on <inline-formula><mml:math id="ieqn73"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mi>M</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="ieqn74"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>U</mml:mi><mml:mi>C</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>.</p><p>The effects are less consistent for FASTopic. General rephrasing improves <inline-formula><mml:math id="ieqn75"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mi>M</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="ieqn76"><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>S</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> in several settings, while colloquial-to-formal rephrasing does not consistently improve <inline-formula><mml:math id="ieqn77"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mi>M</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> relative to the nonrephrased baseline. This suggests that embedding-based or neural topic models may already capture some semantic relationships in noisy short texts, making them less sensitive to input-level formalization than LDA. Therefore, these results support a more cautious interpretation: TM-Rephrase generally improves topic coherence, especially for LDA, BERTopic, and TSCTM, but the magnitude of effects is topic model-dependent.</p></sec><sec id="s3-2-6"><title>Sensitivity Analysis Across Different Topic Numbers</title><p>As shown in <xref ref-type="fig" rid="figure4">Figure 4</xref>, both general and colloquial-to-formal rephrasing improved <inline-formula><mml:math id="ieqn78"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> coherence across all tested K values for both models. For BERTopic, colloquial-to-formal rephrasing improved <inline-formula><mml:math id="ieqn79"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> from 0.3445 to 0.3741 at 5 and from 0.4096 to 0.4682 at 20. For LDA, colloquial-to-formal rephrasing improved <inline-formula><mml:math id="ieqn80"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> from 0.3102 to 0.4093 at 5 and from 0.3411 to 0.4002 at 20. These results suggest that the coherence gains from TM-Rephrase are not limited to the original 8 setting, although the magnitude of improvement varies across topic numbers and topic models.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Sensitivity of topic coherence to number of topics and rephrasing schemes for (A) BERTopic and (B) LDA, based on Gemini 2.5 Flash rephrasing backbone. BERTopic: BERT-based topic modeling; LDA: latent Dirichlet allocation.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="infodemiology_v6i1e98319_fig08.png"/></fig><p>These quantitative results suggest that TM-Rephrase generally enhances intratopic coherence and can improve intertopic quality metrics for short, noisy texts in health communications, although the effects are model- and metric-dependent. The 2 different rephrasing schemes that we developed have similar overall performance. The general rephrasing scheme offers balanced and reliable improvements, while the colloquial-to-formal scheme has a slight edge, especially for lexically sensitive topic models such as LDA and BERTopic. These findings demonstrate the effectiveness and robustness of TM-Rephrase as an LLM-based model-agnostic framework for topic modeling in health communications.</p></sec></sec><sec id="s3-3"><title>Results of Expert Ratings of Rephrasing Semantic Fidelity</title><p>The expert validation results (<xref ref-type="table" rid="table4">Table 4</xref>) provided additional evidence regarding the semantic fidelity of the 2 LLM-based rephrasing schemes. Across all 240 expert ratings, the mean semantic fidelity score was 3.90 (SD 1.01), and the median score was 4.00 (IQR 3.00-5.00) out of 5.0, suggesting that rephrasing generally preserved the original meaning.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Results of expert validation on rephrasing semantic fidelity.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Validation scope</td><td align="left" valign="bottom">Ratings</td><td align="left" valign="bottom">Mean (SD)</td><td align="left" valign="bottom">Median (IQR)</td><td align="left" valign="bottom">Ratings &#x2265;4 (%)</td><td align="left" valign="bottom">ICC(2,4)<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup>, average-measure</td></tr></thead><tbody><tr><td align="left" valign="top">Overall</td><td align="left" valign="top">240</td><td align="left" valign="top">3.90 (1.01)</td><td align="left" valign="top">4.00 (3.00-5.00)</td><td align="left" valign="top">65.42</td><td align="left" valign="top">0.736</td></tr><tr><td align="left" valign="top">General rephrasing</td><td align="left" valign="top">120</td><td align="left" valign="top">4.38 (0.78)</td><td align="left" valign="top">5.00 (4.00-5.00)</td><td align="left" valign="top">87.50</td><td align="left" valign="top">0.613</td></tr><tr><td align="left" valign="top">C-to-f<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup> rephrasing</td><td align="left" valign="top">120</td><td align="left" valign="top">3.42 (0.99)</td><td align="left" valign="top">3.00 (3.00-4.00)</td><td align="left" valign="top">43.33</td><td align="left" valign="top">0.553</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>ICC: intraclass correlation coefficient.</p></fn><fn id="table4fn2"><p><sup>b</sup>C-to-f: colloquial-to-formal.</p></fn></table-wrap-foot></table-wrap><p>However, the 2 rephrasing schemes showed different fidelity patterns. General rephrasing received higher expert ratings, with a mean score of 4.38 (SD 0.78), a median score of 5.00 (IQR 4.00-5.00), and 87.50% (105/120) of ratings of 4 or above. In contrast, the colloquial-to-formal scheme received a lower mean score of 3.42 (SD 0.99), a median score of 3.00 (IQR 3.00-4.00), and 43.33% (52/120) of ratings of 4 or above. This result suggests that general rephrasing better preserved the original meaning, stance, or intent, and tone, whereas colloquial-to-formal rephrasing introduced greater changes because of its stronger formalization.</p><p>As meaning, stance, or intent, and tone were evaluated through a single holistic rating rather than separate dimension-specific scores, the validation results cannot determine quantitatively which aspect of fidelity contributed most to the lower ratings for colloquial-to-formal rephrasing. Therefore, the lower colloquial-to-formal ratings should be interpreted as indicating a greater risk of fidelity change overall, rather than as definitive evidence that tone or stance or intent alone was the primary source of reduced fidelity. However, 1 expert noted that meaning and stance or intent were often similar while tone differed, suggesting that professionalization of tone may be one important contributor.</p><p>The overall average interrater reliability was ICC(2,4)=0.736, indicating acceptable agreement among expert raters for the aggregated semantic fidelity ratings. Scheme-specific intraclass correlation coefficient values were lower, with ICC(2,4)=0.613 for general rephrasing and ICC(2,4)=0.553 for colloquial-to-formal rephrasing. This pattern suggests that expert judgments were more variable when evaluating the degree to which formalized rephrasing preserved tone and stance, which is consistent with the greater linguistic transformation involved in the colloquial-to-formal scheme. Generally, these findings support a more nuanced interpretation: general rephrasing provides stronger preservation of the original public voice, whereas colloquial-to-formal rephrasing offers clearer formalization but involves a possible trade-off in tone and rhetorical fidelity.</p></sec><sec id="s3-4"><title>Illustrative Topic-Level Examples</title><p>To contextualize the quantitative findings, <xref ref-type="table" rid="table5">Tables 5</xref> and <xref ref-type="table" rid="table6">6</xref> provide illustrative examples of topic keywords and topic assignments before and after rephrasing.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Top 15 keywords for LDA<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup> topics based on Gemini 2.5 Flash (<italic>K</italic>=8)<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup>.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Topic ID</td><td align="left" valign="bottom">Keywords</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">w/o<sup><xref ref-type="table-fn" rid="table5fn3">c</xref></sup> rephrasing</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>1</td><td align="left" valign="top">vaccine, covid, get, people, dont<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, mask, kid, cdc, child, need, school, vaccinated, like<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, work, trump</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>2</td><td align="left" valign="top">mask, covid, wear, virus, vaccine, wearing, people, stop, dont<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, flu, need, face, work, social, spread</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>3</td><td align="left" valign="top">covid, death, vaccine, cdc, people, case, day<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, many<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, child, one<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, dont,<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup> symptom, week<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, died, year<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>4</td><td align="left" valign="top">vaccine, effect, dose, people, covid, know, covaxin, dont<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, pfizer, shot, like<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, mrna, child, long, one<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>5</td><td align="left" valign="top">covid, death, rate, case, cdc, data, infection, people, immunity, vaccination, study, show<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup><underline>,</underline> variant, state, number</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>6</td><td align="left" valign="top">vaccine, cdc, covid, pfizer, stop, fda, people, transmission, please<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, health, year<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, child, approved, public, prevent</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>7</td><td align="left" valign="top">covid, ivermectin, dont<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, mask, cdc, people, work, vaccine, stop, positive, tested, like<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, getting<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, dose, doctor</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>8</td><td align="left" valign="top">vaccine, covid, vaccinated, get<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, flu, still<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, shot, people, child, year<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, got<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, even<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, booster, fully<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, getting<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup></td></tr><tr><td align="left" valign="top" colspan="2">w/<sup><xref ref-type="table-fn" rid="table5fn5">e</xref></sup> general rephrasing scheme</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>1</td><td align="left" valign="top">covid, cdc, vaccine, death, please<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, mask, pandemic, trump, virus, could<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, american, health, public, regarding, trust</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>2</td><td align="left" valign="top">mask, vaccinate, wear, covid, please<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, personal_protective_equipment, get<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, vaccine, school, individual, society, social_distance, virus, wearing, still<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>3</td><td align="left" valign="top">covid, vaccine, individual, flu, vaccinated, people, case, virus, immunity, child, stop, positive, tested, get<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, one<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>4</td><td align="left" valign="top">vaccine, covid, ivermectin, mrna, pfizer, people, lie, individual, death, effective, health, covaxin, received, moderna, treatment</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>5</td><td align="left" valign="top">vaccine, covid, child, shot, booster, effective, mrna, prevent, flu, please, receive, cdc, people, covaxin, variant</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>6</td><td align="left" valign="top">mask, covid, day<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, cases, test, people, virus, individual, cdc, one<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, work, vaccine, hospital, positive, wear</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>7</td><td align="left" valign="top">covid, child, vaccine, death, risk, vaccination, data, year<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, people, cdc, rate, individual, adverse, virus, cause</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>8</td><td align="left" valign="top">vaccine, risk, side_effect, testing, tweet, pfizer, would<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, covid, child, vaccination, kid, also<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup><underline>,</underline> hospital, concern, original</td></tr><tr><td align="left" valign="top" colspan="2">w/ c-to-f<sup><xref ref-type="table-fn" rid="table5fn6">f</xref></sup> rephrasing scheme</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>1</td><td align="left" valign="top">mask, covid, individual, vaccine, public, operation, vaccinated, state, virus, health, wear, public_measure, may, work, mandate</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>2</td><td align="left" valign="top">covid, child, vaccine, individual, vaccination, regarding, death, concern, risk, year<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, virus, age, variant, may, significant</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>3</td><td align="left" valign="top">covid, individual, vaccine, positive, test, statement, author, mask, vaccination, testing, result, current, case, information, president</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>4</td><td align="left" valign="top">disease, cdc, prevention, control, center, covid, tweet, ivermectin, individual, health, professional, public, treatment, travel, coronavirus</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>5</td><td align="left" valign="top">vaccine, covid, effect, associated, data, symptom, adverse, cdc, child, efficiency, vaccination, side_effect, potential, disease, death</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>6</td><td align="left" valign="top">individual, vaccine, vaccination, covid, vaccinated, regarding, may, immunity, infection, health, efficacy, influenza, public, risk, concern</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>7</td><td align="left" valign="top">vaccine, vaccination, covid, individual, booster, pfizer, received, pharmaceutical, dose, trust, cdc, company, fda, financial, administration</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>8</td><td align="left" valign="top">public, health, regarding, covid, concern, pandemic, vaccine, social_distance, significant, information, current, mask, trump, cdc, individual</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>LDA: latent Dirichlet allocation.</p></fn><fn id="table5fn2"><p><sup>b</sup>Results with rephrasing are based on the input texts rephrased through Gemini 2.5 Flash.</p></fn><fn id="table5fn3"><p><sup>c</sup>w/o: without.</p></fn><fn id="table5fn4"><p><sup>d</sup>Irrelevant keywords (off-topic words identified through author interpretation).</p></fn><fn id="table5fn5"><p><sup>e</sup>w/: with.</p></fn><fn id="table5fn6"><p><sup>f</sup>C-to-f: colloquial-to-formal.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Illustrative examples of original, general, and c-to-f<sup><xref ref-type="table-fn" rid="table6fn1">a</xref></sup> rephrased tweet replies with corresponding LDA<sup><xref ref-type="table-fn" rid="table6fn2">b</xref></sup> and BERTopic<sup><xref ref-type="table-fn" rid="table6fn3">c</xref></sup> topic assignments<sup><xref ref-type="table-fn" rid="table6fn4">d</xref></sup>.</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Text type</td><td align="left" valign="bottom">Text content</td><td align="left" valign="bottom">Topic assigned by LDA</td><td align="left" valign="bottom">Topic assigned by BERTopic</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="4">Example 1</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Original</td><td align="left" valign="top">Are you kidding me with this? Don&#x2019;t be stupid enough to let your perfect baby be a guinea pig for these people. People are getting bells palsy after getting the vaccine.</td><td align="left" valign="top">vaccine<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, covid, get, people, dont, mask, kid<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, cdc, child<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, need, school, vaccinated<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, like, work, trump</td><td align="left" valign="top">covaxin, novavax, vaccine<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, approve, mrna, booster, ichoosecovaxin, need, want, fda, safe<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, people, child<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, wait, ocugen</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>General rephrased</td><td align="left" valign="top">Are you serious? Don&#x2019;t be foolish and allow your healthy child to be a test subject for these individuals. Reports indicate people are developing Bell&#x2019;s palsy after receiving the vaccine.</td><td align="left" valign="top">vaccine<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, risk<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, side_effect<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, testing<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, tweet, pfizer, would, covid, child<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, vaccination<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, kid<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, also, hospital, concern<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup><underline>,</underline> original</td><td align="left" valign="top">vaccine<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, covaxin, omicron, novavax, variant, mrna, delta, booster, approve, child<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, vaccinate, side_effect<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, people, risk<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, concern<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>C-to-f rephrased</td><td align="left" valign="top">Concerns have been raised regarding potential adverse effects, specifically Bell&#x2019;s palsy, following vaccination. It is imperative to approach public health recommendations and medical interventions with informed decision-making.</td><td align="left" valign="top">vaccine<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, covid, effect, associated, data, symptom, adverse<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, cdc, child<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, efficiency, vaccination<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, side_effect<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup><underline>, </underline>potential, disease, death</td><td align="left" valign="top">vaccine<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, covaxin, novavax, mrna, approval, subject<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, receive, adverse<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, booster, vaccination, risk<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, efficacy, child<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, side_effect<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, concern<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup></td></tr><tr><td align="left" valign="top" colspan="4">Example 2.</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Original</td><td align="left" valign="top">Yes - let&#x2019;s trust Pfizer - same company that entered a $2.3 billion criminal plea deal for lying! Same company that is exempt from liability and CANT be sued for any injuries - yup sign me right up *eye roll*</td><td align="left" valign="top">vaccine<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, effect, dose, people, covid, know, covaxin, dont, pfizer<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, shot, like, mrna, child, long, one</td><td align="left" valign="top">pfizer<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, vaccine<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, effect, mrna, covid, cdc, moderna, people, shot, child, test, get, know, stop, receive</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>General rephrased</td><td align="left" valign="top">Should we trust Pfizer? The same company that previously entered a $2.3 billion criminal plea deal for lying is now exempt from liability and cannot be sued for any injuries. So, yes, sign me right up. *eye roll*</td><td align="left" valign="top">vaccine<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, covid, ivermectin, mrna, pfizer<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, people, lie<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup><underline>,</underline> individual, death, effective, health, covaxin, received, moderna, treatment</td><td align="left" valign="top">vaccine<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, pfizer<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, individual, vaccination, trial, mrna, concern<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, lie<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, report, utility, public, moderna, cdc, fda, information</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>C-to-f rephrased</td><td align="left" valign="top">Concerns exist regarding public trust in Pfizer, given its history, including a $2.3 billion criminal plea deal for deceptive practices. Furthermore, the company&#x2019;s exemption from liability and inability to be sued for injuries raises significant questions regarding accountability.</td><td align="left" valign="top">vaccine<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, vaccination<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, covid, individual, booster, pfizer<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, received, pharmaceutical, dose, trust<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, cdc, company<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, fda, financial, administration</td><td align="left" valign="top">vaccine<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, pfizer<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, individual, vaccination<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, responsibility<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, mrna, concern<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, profit<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, business, trust<sup><xref ref-type="table-fn" rid="table6fn5">e</xref></sup>, public, moderna, cdc, fda, information</td></tr></tbody></table><table-wrap-foot><fn id="table6fn1"><p><sup>a</sup>C-to-f: colloquial-to-formal.</p></fn><fn id="table6fn2"><p><sup>b</sup>LDA: latent Dirichlet allocation. </p></fn><fn id="table6fn3"><p><sup>c</sup>BERTopic: BERT-based topic modeling.</p></fn><fn id="table6fn4"><p><sup>d</sup>Results with rephrasing are based on the input texts rephrased through Gemini 2.5 Flash.</p></fn><fn id="table6fn5"><p><sup>e</sup>Words that better describe topics.</p></fn></table-wrap-foot></table-wrap><p>These examples are intended to demonstrate how TM-Rephrase may affect interpretability. In interpreting these examples, keywords were considered more informative when they were specific, domain-relevant, and clearly aligned with public health themes, whereas generic terms, colloquial artifacts, and weakly related words were treated as less informative.</p><p>In <xref ref-type="table" rid="table5">Table 5</xref>, LDA-derived topics without rephrasing exhibit limited semantic coherence, particularly for topic 2 (public health measures), which is dominated by generic and less informative terms (eg, &#x201C;mask,&#x201D; &#x201C;wear,&#x201D; and &#x201C;stop&#x201D;) and includes off-topic noise such as &#x201C;dont.&#x201D; General rephrasing improves topical specificity by introducing more contextually meaningful health-related terms (eg, &#x201C;personal_protective_equipment&#x201D; and &#x201C;social_distance&#x201D;), although some colloquial artifacts persist (eg, &#x201C;please&#x201D; and &#x201C;could&#x201D;).</p><p>In contrast, colloquial-to-formal rephrasing produces more coherent and structured topics, characterized by standardized and policy-relevant terminology (eg, &#x201C;mandate&#x201D; and &#x201C;public_measure&#x201D;). A similar pattern is observed in vaccine-related topics (eg, topic 5), where baseline outputs contain vague terms (eg, &#x201C;like&#x201D; and &#x201C;get&#x201D;), while rephrasing, particularly for colloquial-to-formal, yields more precise and analytically meaningful keywords (eg, &#x201C;adverse,&#x201D; &#x201C;associated,&#x201D; and &#x201C;efficiency&#x201D;). These examples suggest that increasing linguistic formality and contextual explicitness can reduce less informative terms and make selected topic outputs easier to interpret in relation to public health discourse.</p><p><xref ref-type="table" rid="table6">Table 6</xref> further illustrates how rephrasing may affect the apparent semantic alignment between selected replies and their assigned topics. In <xref ref-type="table" rid="table6">Table 6</xref>, rephrasing improves the semantic alignment between tweet content and assigned topics across both LDA and BERTopic. In example 1, the original tweet expresses concerns about vaccine side effects using informal and fragmented language, resulting in generic or weakly focused topic keywords (eg, &#x201C;vaccine&#x201D; and &#x201C;child&#x201D;). General rephrasing introduces explicit health-related terms such as &#x201C;side effect,&#x201D; &#x201C;risk,&#x201D; and &#x201C;concern,&#x201D; improving thematic alignment. Colloquial-to-formal rephrasing further strengthens this alignment by incorporating more formal and domain-specific terminology (eg, &#x201C;adverse&#x201D;), enabling both models to produce more coherent and semantically grounded topics. A similar pattern is observed in example 2, where the original tweet contains colloquial and emotionally charged expressions regarding Pfizer.</p><p>Topic assignments from the original text include noisy and less informative terms (eg, &#x201C;people,&#x201D; &#x201C;dont,&#x201D; and &#x201C;get&#x201D;). General rephrasing improves relevance by introducing clearer semantic cues (eg, &#x201C;concern&#x201D;), while colloquial-to-formal rephrasing further refines the discourse with more structured and policy-relevant vocabulary (eg, &#x201C;trust,&#x201D; &#x201C;responsibility,&#x201D; and &#x201C;profit&#x201D;), allowing topic models, particularly BERTopic, to capture higher-level institutional themes.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings<inline-graphic xlink:href="infodemiology_v6i1e98319_fig09.png"/></title><p>This study developed and evaluated TM-Rephrase, a model-agnostic framework that uses LLM-based rephrasing to standardize short, informal public health texts while aiming to preserve their original meaning. Using 25,027 public replies to CDC tweets on Twitter during the COVID-19 pandemic, we found that rephrasing improved topic coherence and topic distinctiveness while generally preserving semantic fidelity across multiple topic modeling approaches.</p><p>For RQ1 (whether and how effectively TM-Rephrase improves the quality and interpretability of topic model outputs for short texts in health communications), the results show that TM-Rephrase improves topic quality and interpretability for short, noisy social media texts. The strongest coherence gains were observed for LDA, where <inline-formula><mml:math id="ieqn81"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> increased from 0.3094 to 0.5004, suggesting that rephrasing helps mitigate lexical sparsity and fragmented wording. Improvements were also observed for embedding-based approaches such as BERTopic, indicating that the benefits of rephrasing are not limited to one type of topic model. Across multiple evaluation metrics, including <inline-formula><mml:math id="ieqn82"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, <inline-formula><mml:math id="ieqn83"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mi>P</mml:mi><mml:mi>M</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, <inline-formula><mml:math id="ieqn84"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>U</mml:mi><mml:mi>C</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, <inline-formula><mml:math id="ieqn85"><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>S</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, <italic>TU</italic>, <italic>TR</italic>, and <italic>TD</italic>, the results suggest that TM-Rephrase produces topic-word distributions that are more coherent, less redundant, and easier to interpret.</p><p>However, these improvements should be interpreted cautiously. Automated coherence metrics may partly reward the lexical regularity introduced by rephrasing, and higher topic coherence does not necessarily mean that all meaningful public-response signals are preserved. In infoveillance contexts, sarcasm, distrust, anger, uncertainty, and oppositional framing are not merely noise; they may represent important public health signals. Therefore, we interpret TM-Rephrase as improving the readability and interpretability of topic-model outputs, rather than as fully validating the substantive meaning of public responses.</p><p>For RQ2 (how different rephrasing schemes influence topic model performance), the 2 rephrasing schemes showed different effects. General rephrasing produced balanced and consistent improvements while better preserving the original public voice. In contrast, the colloquial-to-formal scheme often produced stronger gains in coherence and topic distinctiveness, especially for lexically sensitive models such as LDA and TSCTM. At the same time, colloquial-to-formal rephrasing involves greater linguistic transformation and may alter tone, rhetorical force, or framing. This suggests that prompt design should be aligned with the analytic goal: general rephrasing may be preferable when preserving public expression is important, while colloquial-to-formal rephrasing may be useful when the goal is to maximize topical clarity and model interpretability.</p><p>For RQ3 (how to interpret public discussions, that is, replies to CDC during the COVID-19 pandemic, based on the topics and inform public health practice), the improved topic representations helped clarify public responses to CDC communications. Rephrased topics more clearly captured recurring concerns related to vaccination attitudes, perceived risks, preventive behaviors, mistrust of official guidance, and uncertainty about public health recommendations. These findings suggest that TM-Rephrase can support public health social listening by helping analysts organize large volumes of noisy public replies into more interpretable thematic structures.</p><p>Practically, TM-Rephrase may serve as a human-in-the-loop preprocessing strategy for public health social listening rather than a replacement for expert interpretation. By reducing lexical fragmentation and improving the readability of topic-model outputs, it may help analysts organize large volumes of noisy public replies and identify themes that warrant closer review, such as vaccine safety concerns, mistrust of official guidance, confusion about preventive measures, or perceived gaps between public messaging and lived experience. However, this study does not directly evaluate whether TM-Rephrase improves public health decisions, communication outcomes, or detection of otherwise missed concerns. Therefore, its public health utility should be interpreted as a potential application that requires further validation in real-world infoveillance and communication workflows.</p><p>Generally, this study provides evidence that LLM-based rephrasing can enhance topic modeling of short public health texts, while also highlighting the need for careful validation. Future work should further evaluate semantic fidelity through larger-scale human expert review, topic intrusion tasks, posttopic alignment assessments, and cross-platform testing across different public health contexts.</p></sec><sec id="s4-2"><title>Limitations and Future Work</title><p>This study has several limitations that should be addressed in future research. First, during dataset curation, we did not perform additional bot or spam classification beyond the filtering described in Data Collection, because the primary objective was to evaluate the effect of rephrasing on topic modeling outputs rather than to infer public opinion. The filtering process may also introduce selection bias. In particular, replies linked to deleted or inaccessible CDC tweets were excluded, and these excluded interactions may differ systematically from the retained replies. Therefore, the final corpus should be interpreted as a dataset of public replies with recoverable CDC conversational context, rather than as a complete representation of all public interactions with CDC accounts or all COVID-19 discourse on Twitter. In addition, because the analysis focuses on replies to official CDC communications on Twitter during the COVID-19 pandemic, future research should test TM-Rephrase across other platforms, health topics, population groups, and public health communication contexts before applying it in operational surveillance or social listening systems.</p><p>Second, although LLM-based rephrasing is designed to preserve semantic meaning while improving linguistic clarity, it may introduce an important trade-off for public health infoveillance. Rephrasing can reduce lexical fragmentation and improve topic coherence, but it may also soften or alter meaningful public-discourse signals, including sarcasm, anger, distrust, misinformation cues, uncertainty, oppositional framing, and colloquial expressions. These features should not be treated merely as noise, because they may reflect public resistance, confusion, risk perception, or distrust toward health institutions.</p><p>The expert semantic-fidelity validation provides partial evidence regarding this trade-off. Overall, expert ratings suggested generally favorable preservation of meaning, stance or intent, and tone; however, general rephrasing received stronger semantic-fidelity ratings than colloquial-to-formal rephrasing. This suggests that stronger formalization may be more likely to alter tone, rhetorical force, or stance expression. Therefore, automated improvements in coherence should not be interpreted as definitive evidence of improved substantive interpretability. In infoveillance applications, original and rephrased texts should be examined together, especially when sarcasm, distrust, anger, uncertainty, misinformation cues, or oppositional framing are central to the RQ. In addition, the expert validation used a single holistic semantic-fidelity rating that asked raters to consider meaning, stance or intent, and tone together; therefore, this study cannot quantitatively determine which specific dimension of fidelity was most affected by rephrasing. Future validation should rate meaning, stance or intent, tone, uncertainty, sarcasm, and distrust as separate dimensions to better identify which aspects of public discourse are most vulnerable to meaning drift or tone normalization.</p><p>Recent AI alignment research suggests that differences across LLM backbones may extend beyond linguistic performance. Lau et al [<xref ref-type="bibr" rid="ref34">34</xref>] found that LLMs can vary in value-priority profiles and alignment with human judgments. This is relevant to TM-Rephrase because rephrasing public discourse may involve implicit choices about tone, framing, emphasis, and socially sensitive meanings. Thus, although our results show broadly similar topic-modeling improvements across the LLMs tested, future work should examine whether different LLMs preserve meaning, stance, tone, uncertainty, and distrust in comparable ways, especially for sensitive public-health replies.</p><p>Moreover, this study used a fixed number of topics and a selected set of representative topic modeling approaches to enable consistent comparison between original and rephrased texts. Although this design supports methodological evaluation, public health communication is dynamic and context-dependent. Future work could adapt TM-Rephrase to better capture evolving public concerns across different stages of health emergencies.</p></sec><sec id="s4-3"><title>Conclusions</title><p>This study highlights the value of improving input text quality for online public health infoveillance and analytics, particularly those based on social media. By leveraging LLM-based rephrasing, TM-Rephrase enables more interpretable, consistent, and reliable detection of thematic patterns in short, noisy public responses to official health communications during health emergencies.</p><p>The data-centric and model-agnostic TM-Rephrase framework also offers a practical and scalable solution to enhance the usability of large-scale social media data for infoveillance. The findings suggest that linguistic standardization plays an important role in strengthening the interpretability of analytical outputs and supporting a clearer understanding of public concerns from large volumes of raw online data.</p><p>As social media continues to serve as a key channel for public engagement on various health topics, novel analytical frameworks, such as TM-Rephrase, enable more effective monitoring of public discourse and better-informed public health communication strategies, especially during health emergencies.</p></sec></sec></body><back><ack><p>The authors gratefully acknowledge Dr Sijia Qian and Dr Ron Lunsford for participating as expert validators in the semantic fidelity assessment of original and large language model&#x2013;rephrased social media tweets. Their feedback helped evaluate whether the rephrased versions preserved the meaning, stance or intent, and tone of the original tweets. The authors declare the use of generative AI (GenAI) in the research and writing process. According to the GAIDeT (Generative AI Delegation Taxonomy [<xref ref-type="bibr" rid="ref35">35</xref>]; 2025), the following tasks were delegated to GenAI tools under full human supervision: selection of research methods, proofreading and editing, and reformatting. The GenAI tool used was ChatGPT-5.5 (OpenAI). Responsibility for the final manuscript lies entirely with the authors. GenAI tools are not listed as authors and do not bear responsibility for the outcomes.</p></ack><notes><sec><title>Funding</title><p>This study was supported by the National Science Foundation (DMS-2436227). The funding organization had no role in this study's design, data collection and analysis, decision to publish, or preparation of this paper. The opinions, findings, conclusions, and recommendations presented in this paper are those of the authors alone and do not necessarily represent the views of the funding organization.</p></sec><sec><title>Data Availability</title><p>The datasets analyzed during this study consist of public replies to Centers for Disease Control and Prevention communications on Twitter (subsequently rebranded as X). The raw tweet-level social media datasets are not publicly shared because redistribution of user-generated social media content on Brandwatch may be restricted by platform terms and because public tweets may still contain content that could raise privacy or reidentification concerns. Aggregated results and derived materials supporting the findings are available from the corresponding author upon reasonable request, subject to applicable platform policies and ethical considerations.</p></sec></notes><fn-group><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">BERTopic</term><def><p>BERT-based topic modeling</p></def></def-item><def-item><term id="abb2">CDC</term><def><p>Centers for Disease Control and Prevention</p></def></def-item><def-item><term id="abb3">FASTopic</term><def><p>fast, adaptive, stable, and transferable topic model</p></def></def-item><def-item><term id="abb4">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb5">LDA</term><def><p>latent Dirichlet allocation</p></def></def-item><def-item><term id="abb6">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb7">NPMI</term><def><p>normalized pointwise mutual information</p></def></def-item><def-item><term id="abb8">RQ</term><def><p>research question</p></def></def-item><def-item><term id="abb9">TD</term><def><p>topic diversity</p></def></def-item><def-item><term id="abb10">TR</term><def><p>topic redundancy</p></def></def-item><def-item><term id="abb11">TSCTM</term><def><p>topic-semantic contrastive topic model</p></def></def-item><def-item><term id="abb12">TU</term><def><p>topic uniqueness</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rauchfleisch</surname><given-names>A</given-names> </name><name name-style="western"><surname>Vogler</surname><given-names>D</given-names> </name><name name-style="western"><surname>Eisenegger</surname><given-names>M</given-names> </name></person-group><article-title>Public sphere in crisis mode: how the COVID-19 pandemic influenced public discourse and user behaviour in the Swiss Twitter-sphere</article-title><source>Javnost</source><year>2021</year><volume>28</volume><issue>2</issue><fpage>129</fpage><lpage>148</lpage><pub-id pub-id-type="doi">10.1080/13183222.2021.1923622</pub-id><pub-id pub-id-type="medline">34393592</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>S</given-names> </name></person-group><article-title>Understanding concerns, sentiments, and disparities among population groups during the COVID-19 pandemic via Twitter data mining: large-scale cross-sectional study</article-title><source>J Med Internet Res</source><year>2021</year><month>03</month><day>5</day><volume>23</volume><issue>3</issue><fpage>e26482</fpage><pub-id pub-id-type="doi">10.2196/26482</pub-id><pub-id pub-id-type="medline">33617460</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yin</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ge</surname><given-names>Y</given-names> </name></person-group><article-title>Dynamic associations between Centers for Disease Control and Prevention social media contents and epidemic measures during COVID-19: infoveillance study</article-title><source>JMIR Infodemiology</source><year>2024</year><month>01</month><day>23</day><volume>4</volume><issue>1</issue><fpage>e49756</fpage><pub-id pub-id-type="doi">10.2196/49756</pub-id><pub-id pub-id-type="medline">38261367</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="web"><article-title>Timeline of X</article-title><source>Wikipedia</source><access-date>2026-09-10</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://en.wikipedia.org/wiki/Timeline_of_Twitter">https://en.wikipedia.org/wiki/Timeline_of_Twitter</ext-link></comment></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Laureate</surname><given-names>CDP</given-names> </name><name name-style="western"><surname>Buntine</surname><given-names>W</given-names> </name><name name-style="western"><surname>Linger</surname><given-names>H</given-names> </name></person-group><article-title>A systematic review of the use of topic models for short text social media analysis</article-title><source>Artif Intell Rev</source><year>2023</year><month>05</month><day>1</day><volume>56</volume><fpage>1</fpage><lpage>33</lpage><pub-id pub-id-type="doi">10.1007/s10462-023-10471-x</pub-id><pub-id pub-id-type="medline">37362887</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Blei</surname><given-names>DM</given-names> </name><name name-style="western"><surname>Ng</surname><given-names>AY</given-names> </name><name name-style="western"><surname>Jordan</surname><given-names>MI</given-names> </name></person-group><article-title>Latent Dirichlet allocation</article-title><source>J Mach Learn Res</source><year>2003</year><access-date>2026-09-10</access-date><volume>3</volume><fpage>993</fpage><lpage>1022</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://www.jmlr.org/papers/volume3/blei03a/blei03a.pdf">https://www.jmlr.org/papers/volume3/blei03a/blei03a.pdf</ext-link></comment></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Hong</surname><given-names>L</given-names> </name><name name-style="western"><surname>Davison</surname><given-names>BD</given-names> </name></person-group><article-title>Empirical study of topic modeling in Twitter</article-title><conf-name>Proceedings of the First Workshop on Social Media Analytics</conf-name><conf-date>Jul 25-28, 2010</conf-date><conf-loc>Washington, DC</conf-loc><fpage>80</fpage><lpage>88</lpage><pub-id pub-id-type="doi">10.1145/1964858.1964870</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Rajagopal</surname><given-names>D</given-names> </name><name name-style="western"><surname>Olsher</surname><given-names>D</given-names> </name><name name-style="western"><surname>Cambria</surname><given-names>E</given-names> </name><name name-style="western"><surname>Kwok</surname><given-names>K</given-names> </name></person-group><article-title>Commonsense-based topic modeling</article-title><conf-name>Proceedings of the second international workshop on issues of sentiment discovery and opinion mining</conf-name><conf-date>Aug 11-12, 2013</conf-date><conf-loc>Chicago, IL</conf-loc><fpage>1</fpage><lpage>8</lpage><pub-id pub-id-type="doi">10.1145/2502069.2502075</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="web"><source>BERTopic</source><access-date>2026-09-10</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://maartengr.github.io/BERTopic/">https://maartengr.github.io/BERTopic/</ext-link></comment></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Angelov</surname><given-names>D</given-names> </name></person-group><article-title>Top2Vec: distributed representations of topics</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 19, 2020</comment><pub-id pub-id-type="doi">10.48550/arXiv.2008.09470</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Nguyen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>WY</given-names> </name><name name-style="western"><surname>Luu</surname><given-names>AT</given-names> </name></person-group><article-title>FASTopic: pretrained transformer is a fast, adaptive, stable, and transferable topic model</article-title><conf-name>Advances in Neural Information Processing Systems 37</conf-name><conf-date>Dec 10-15, 2024</conf-date><conf-loc>Vancouver, BC, Canada</conf-loc><fpage>84447</fpage><lpage>84481</lpage><pub-id pub-id-type="doi">10.52202/079017-2683</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Luu</surname><given-names>AT</given-names> </name><name name-style="western"><surname>Dong</surname><given-names>X</given-names> </name></person-group><article-title>Mitigating data sparsity for short text topic modeling by topic-semantic contrastive learning</article-title><conf-name>Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Dec 7-11, 2022</conf-date><conf-loc>Abu Dhabi, United Arab Emirates</conf-loc><fpage>2748</fpage><lpage>2760</lpage><pub-id pub-id-type="doi">10.18653/v1/2022.emnlp-main.176</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>JZ</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Wong</surname><given-names>JC</given-names> </name></person-group><article-title>Information seeking and information sharing during the COVID-19 pandemic</article-title><source>Commun Q</source><year>2022</year><month>01</month><day>1</day><volume>70</volume><issue>1</issue><fpage>1</fpage><lpage>21</lpage><pub-id pub-id-type="doi">10.1080/01463373.2021.1995772</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>AbuRaed</surname><given-names>AGT</given-names> </name><name name-style="western"><surname>Prikryl</surname><given-names>EA</given-names> </name><name name-style="western"><surname>Carenini</surname><given-names>G</given-names> </name><name name-style="western"><surname>Janjua</surname><given-names>NZ</given-names> </name></person-group><article-title>Long COVID Discourse in Canada, the United States, and Europe: topic modeling and sentiment analysis of Twitter data</article-title><source>J Med Internet Res</source><year>2024</year><month>12</month><day>9</day><volume>26</volume><fpage>e59425</fpage><pub-id pub-id-type="doi">10.2196/59425</pub-id><pub-id pub-id-type="medline">39652387</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>James</surname><given-names>L</given-names> </name><name name-style="western"><surname>McPhail</surname><given-names>H</given-names> </name><name name-style="western"><surname>Foisey</surname><given-names>L</given-names> </name><name name-style="western"><surname>Donelle</surname><given-names>L</given-names> </name><name name-style="western"><surname>Bauer</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kothari</surname><given-names>A</given-names> </name></person-group><article-title>Exploring communication by public health leaders and organizations during the pandemic: a content analysis of COVID-related tweets</article-title><source>Can J Public Health</source><year>2023</year><month>08</month><volume>114</volume><issue>4</issue><fpage>563</fpage><lpage>583</lpage><pub-id pub-id-type="doi">10.17269/s41997-023-00783-4</pub-id><pub-id pub-id-type="medline">37349662</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>E</given-names> </name><name name-style="western"><surname>Lerman</surname><given-names>K</given-names> </name><name name-style="western"><surname>Ferrara</surname><given-names>E</given-names> </name></person-group><article-title>Tracking Social Media Discourse About the COVID-19 Pandemic: Development of a Public Coronavirus Twitter Data Set</article-title><source>JMIR Public Health Surveill</source><year>2020</year><month>05</month><day>29</day><volume>6</volume><issue>2</issue><fpage>e19273</fpage><pub-id pub-id-type="doi">10.2196/19273</pub-id><pub-id pub-id-type="medline">32427106</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Maini</surname><given-names>P</given-names> </name><name name-style="western"><surname>Seto</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bai</surname><given-names>R</given-names> </name><name name-style="western"><surname>Grangier</surname><given-names>D</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Jaitly</surname><given-names>N</given-names> </name></person-group><article-title>Rephrasing the web: a recipe for compute and data-efficient language modeling</article-title><conf-name>Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1</conf-name><conf-loc>Bangkok, Thailand</conf-loc><fpage>14044</fpage><lpage>14072</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.acl-long.757</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>H</given-names> </name><name name-style="western"><surname>Meng</surname><given-names>D</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>J</given-names> </name></person-group><article-title>A comprehensive survey on automatic text summarization with exploration of LLM-based methods</article-title><source>Neurocomputing</source><year>2026</year><month>01</month><volume>663</volume><fpage>131928</fpage><pub-id pub-id-type="doi">10.1016/j.neucom.2025.131928</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Miah</surname><given-names>MSU</given-names> </name><name name-style="western"><surname>Kabir</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Sarwar</surname><given-names>TB</given-names> </name><name name-style="western"><surname>Safran</surname><given-names>M</given-names> </name><name name-style="western"><surname>Alfarhood</surname><given-names>S</given-names> </name><name name-style="western"><surname>Mridha</surname><given-names>MF</given-names> </name></person-group><article-title>A multimodal approach to cross-lingual sentiment analysis with ensemble of transformer and LLM</article-title><source>Sci Rep</source><year>2024</year><month>04</month><day>26</day><volume>14</volume><issue>1</issue><fpage>9603</fpage><pub-id pub-id-type="doi">10.1038/s41598-024-60210-7</pub-id><pub-id pub-id-type="medline">38671064</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Information retrieval meets large language models</article-title><conf-name>Companion Proceedings of the ACM Web Conference 2024</conf-name><conf-date>May 13-17, 2024</conf-date><conf-loc>Singapore, Singapore</conf-loc><fpage>1586</fpage><lpage>1589</lpage><pub-id pub-id-type="doi">10.1145/3589335.3641299</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="web"><article-title>Gemini 25 flash model</article-title><source>Google Cloud</source><access-date>2026-09-10</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://docs.cloud.google.com/vertex-ai/generative-ai/docs/models/gemini/2-5-flash">https://docs.cloud.google.com/vertex-ai/generative-ai/docs/models/gemini/2-5-flash</ext-link></comment></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="web"><article-title>GPT-4o mini: advancing cost-efficient intelligence</article-title><source>OpenAI</source><access-date>2026-09-10</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://openai.com/index/gpt-4o-mini-advancing-cost-efficient-intelligence/">https://openai.com/index/gpt-4o-mini-advancing-cost-efficient-intelligence/</ext-link></comment></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="web"><article-title>Mistral-7B-instruct-v02</article-title><source>Hugging Face</source><access-date>2026-09-10</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://huggingface.co/mistralai/Mistral-7B-Instruct-v0.2">https://huggingface.co/mistralai/Mistral-7B-Instruct-v0.2</ext-link></comment></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Renze</surname><given-names>M</given-names> </name></person-group><article-title>The effect of sampling temperature on problem solving in large language models</article-title><conf-name>Findings of the Association for Computational Linguistics: EMNLP 2024</conf-name><conf-date>Nov 12-16, 2024</conf-date><conf-loc>Miami, FL</conf-loc><fpage>7346</fpage><lpage>7356</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.findings-emnlp.432</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Reimers</surname><given-names>N</given-names> </name><name name-style="western"><surname>Gurevych</surname><given-names>I</given-names> </name></person-group><article-title>Sentence-BERT: sentence embeddings using siamese BERT-networks</article-title><conf-name>Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)</conf-name><conf-loc>Hong Kong, China</conf-loc><fpage>3982</fpage><lpage>3992</lpage><pub-id pub-id-type="doi">10.18653/v1/D19-1410</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>R&#x00F6;der</surname><given-names>M</given-names> </name><name name-style="western"><surname>Both</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hinneburg</surname><given-names>A</given-names> </name></person-group><article-title>Exploring the space of topic coherence measures</article-title><conf-name>Proceedings of the Eighth ACM International Conference on Web Search and Data Mining</conf-name><conf-date>Feb 2-6, 2015</conf-date><pub-id pub-id-type="doi">10.1145/2684822.2685324</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="thesis"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>X</given-names> </name></person-group><article-title>Towards effective neural topic modeling</article-title><year>2024</year><access-date>2026-09-10</access-date><publisher-name>Nanyang Technological University</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://dr.ntu.edu.sg/bitstreams/2b7465e1-7444-4cda-9078-9e7163cd4006/download">https://dr.ntu.edu.sg/bitstreams/2b7465e1-7444-4cda-9078-9e7163cd4006/download</ext-link></comment></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="web"><article-title>English Wikipedia database dump</article-title><source>Wikimedia Foundation</source><access-date>2026-09-10</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://dumps.wikimedia.org/enwiki/latest/">https://dumps.wikimedia.org/enwiki/latest/</ext-link></comment></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Nan</surname><given-names>F</given-names> </name><name name-style="western"><surname>Ding</surname><given-names>R</given-names> </name><name name-style="western"><surname>Nallapati</surname><given-names>R</given-names> </name><name name-style="western"><surname>Xiang</surname><given-names>B</given-names> </name></person-group><article-title>Topic modeling with Wasserstein autoencoders</article-title><conf-name>Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics</conf-name><conf-date>Jul 28 to Aug 2, 2019</conf-date><conf-loc>Florence, Italy</conf-loc><fpage>6345</fpage><lpage>6381</lpage><pub-id pub-id-type="doi">10.18653/v1/P19-1640</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Burkhardt</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kramer</surname><given-names>S</given-names> </name></person-group><article-title>Decoupling sparsity and smoothness in the Dirichlet variational autoencoder topic model</article-title><source>J Mach Learn Res</source><year>2019</year><access-date>2026-09-10</access-date><volume>20</volume><issue>131</issue><fpage>1</fpage><lpage>27</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://jmlr.org/papers/v20/18-569.html">https://jmlr.org/papers/v20/18-569.html</ext-link></comment></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dieng</surname><given-names>AB</given-names> </name><name name-style="western"><surname>Ruiz</surname><given-names>FJR</given-names> </name><name name-style="western"><surname>Blei</surname><given-names>DM</given-names> </name></person-group><article-title>Topic modeling in embedding spaces</article-title><source>Trans Assoc Comput Linguist</source><year>2020</year><month>12</month><volume>8</volume><fpage>439</fpage><lpage>453</lpage><pub-id pub-id-type="doi">10.1162/tacl_a_00325</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Newman</surname><given-names>D</given-names> </name><name name-style="western"><surname>Lau</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Grieser</surname><given-names>K</given-names> </name></person-group><article-title>Automatic evaluation of topic coherence</article-title><access-date>2026-09-10</access-date><conf-name>Human Language Technologies: The 2010 Annual Conference of the North American Chapter of the Association for Computational Linguistics</conf-name><conf-date>Jun 2-4, 2010</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/N10-1012.pdf">https://aclanthology.org/N10-1012.pdf</ext-link></comment></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Mimno</surname><given-names>D</given-names> </name><name name-style="western"><surname>Wallach</surname><given-names>H</given-names> </name><name name-style="western"><surname>Talley</surname><given-names>E</given-names> </name></person-group><article-title>Optimizing semantic coherence in topic models</article-title><conf-name>Proceedings of the 2011 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Jul 27-31, 2011</conf-date><pub-id pub-id-type="doi">10.5555/2145432.2145462</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Lau</surname><given-names>GR</given-names> </name><name name-style="western"><surname>Low</surname><given-names>WY</given-names> </name><name name-style="western"><surname>Koh</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Nah</surname><given-names>FFH</given-names> </name><name name-style="western"><surname>Hartanto</surname><given-names>A</given-names> </name></person-group><article-title>Evaluating AI alignment in LLMs: output analysis of value priorities across 75 models with human benchmarking</article-title><source>arXiv</source><comment>Preprint posted online on  May 16, 2026</comment><pub-id pub-id-type="doi">10.48550/arXiv.2506.12617</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Suchikova</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Tsybuliak</surname><given-names>N</given-names> </name><name name-style="western"><surname>Teixeira da Silva</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Nazarovets</surname><given-names>S</given-names> </name></person-group><article-title>GAIDeT (Generative AI Delegation Taxonomy): A taxonomy for humans to delegate tasks to generative artificial intelligence in scientific research and publishing</article-title><source>Account Res</source><year>2026</year><month>04</month><volume>33</volume><issue>3</issue><fpage>2544331</fpage><pub-id pub-id-type="doi">10.1080/08989621.2025.2544331</pub-id><pub-id pub-id-type="medline">40781729</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Queries, prompts, implementation details, and notice of the institutional review board exemption determination.</p><media xlink:href="infodemiology_v6i1e98319_app1.pdf" xlink:title="PDF File, 363 KB"/></supplementary-material></app-group></back></article>