<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<?covid-19-tdm?>
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Public Health</journal-id>
<journal-title>Frontiers in Public Health</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Public Health</abbrev-journal-title>
<issn pub-type="epub">2296-2565</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpubh.2023.1227807</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Public Health</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>The Long COVID experience from a patient&#x00027;s perspective: a clustering analysis of 27,216 Reddit posts</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Ayadi</surname> <given-names>Hanin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2403336/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Bour</surname> <given-names>Charline</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2403985/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Fischer</surname> <given-names>Aur&#x000E9;lie</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1541789/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Ghoniem</surname> <given-names>Mohammad</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2361008/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Fagherazzi</surname> <given-names>Guy</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2284624/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Deep Digital Phenotyping Research Unit, Department of Precision Health, Luxembourg Institute of Health</institution>, <addr-line>Strassen</addr-line>, <country>Luxembourg</country></aff>
<aff id="aff2"><sup>2</sup><institution>Faculty of Science, Technology and Medicine, University of Luxembourg</institution>, <addr-line>Esch-sur-Alzette</addr-line>, <country>Luxembourg</country></aff>
<aff id="aff3"><sup>3</sup><institution>&#x000C9;cole doctorale Biologie, Sant&#x000E9;, et Environnement, Universit&#x000E9; de Lorraine</institution>, <addr-line>Nancy</addr-line>, <country>France</country></aff>
<aff id="aff4"><sup>4</sup><institution>Luxembourg Institute of Science and Technology</institution>, <addr-line>Esch-sur-Alzette</addr-line>, <country>Luxembourg</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Harpreet Kaur, National Institutes of Health (NIH), United States</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Andreas Stengel, University Hospital T&#x000FC;bingen, Germany; James Danowski, University of Illinois Chicago, United States; Luis Del Carpio-Orantes, Instituto Mexicano del Seguro Social, Delegaci&#x000F3;n Veracruz Norte, Mexico</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Guy Fagherazzi <email>guy.fagherazzi&#x00040;lih.lu</email>; &#x00040;GFaghe</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>17</day>
<month>08</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>11</volume>
<elocation-id>1227807</elocation-id>
<history>
<date date-type="received">
<day>25</day>
<month>05</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>01</day>
<month>08</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2023 Ayadi, Bour, Fischer, Ghoniem and Fagherazzi.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Ayadi, Bour, Fischer, Ghoniem and Fagherazzi</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license> </permissions>
<abstract>
<sec>
<title>Objective</title>
<p>This work aims to study the profiles of Long COVID from the perspective of the patients spontaneously sharing their experiences and symptoms on Reddit.</p>
</sec>
<sec>
<title>Methods</title>
<p>We collected 27,216 posts shared between July 2020 and July 2022 on Long COVID-related Reddit forums. Natural language processing, clustering techniques and a Long COVID symptoms lexicon were used to extract the different symptoms and categories of symptoms and to study the co-occurrences and correlation between them.</p>
</sec>
<sec>
<title>Results</title>
<p>More than 78% of the posts mentioned at least one Long COVID symptom. Fatigue (29.4%), pain (22%), clouded consciousness (19.1%), anxiety (17.7%) and headaches (15.6%) were the most prevalent symptoms. They also highly co-occurred with a variety of other symptoms (e.g., fever, sinonasal congestion). Different categories of symptoms were found: general (45.5%), neurological/ocular (42.9%), mental health/psychological/behavioral (35.2%), body pain/mobility (35.1%) and cardiorespiratory (31.2%). Posts focusing on other concerns of the community such as vaccine, recovery and relapse and, symptom triggers were detected.</p>
</sec>
<sec>
<title>Conclusions</title>
<p>We demonstrated the benefits of leveraging large volumes of data from Reddit to characterize the heterogeneity of Long COVID profiles. General symptoms, particularly fatigue, have been reported to be the most prevalent and frequently co-occurred with other symptoms. Other concerns, such as vaccination and relapse following recovery, were also addressed by the Long COVID community.</p>
</sec></abstract>
<kwd-group>
<kwd>public health</kwd>
<kwd>patient-reported outcomes</kwd>
<kwd>Long COVID</kwd>
<kwd>digital health</kwd>
<kwd>social media</kwd>
<kwd>artificial intelligence</kwd>
<kwd>natural language processing</kwd>
<kwd>machine learning</kwd>
</kwd-group>
<contract-num rid="cn001">16487884</contract-num>
<contract-sponsor id="cn001">Fonds National de la Recherche Luxembourg<named-content content-type="fundref-id">10.13039/501100001866</named-content></contract-sponsor>
<counts>
<fig-count count="7"/>
<table-count count="1"/>
<equation-count count="0"/>
<ref-count count="31"/>
<page-count count="13"/>
<word-count count="6014"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Digital Public Health</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>Introduction</title>
<p>As of November 2022, more than 630 million confirmed COVID-19 cases and 6.5 million consequential deaths have been reported by the WHO (<xref ref-type="bibr" rid="B1">1</xref>). Even though most infected people fully recover from the initial acute illness, at least 10&#x02013;20% of the recovered patients experience mid and long-term effects after the said recovery (<xref ref-type="bibr" rid="B2">2</xref>).</p>
<p>Long COVID, officially referred to as Post-acute sequelae of SARS-CoV-2 infection (PASC), is a condition where people experience a variety of symptoms that persist from or develop after their initial infection with COVID-19 (<xref ref-type="bibr" rid="B3">3</xref>).</p>
<p>Long COVID symptoms are varied, multi-organ and can range from fatigue to cardio-respiratory issues, including cognitive dysfunction (<xref ref-type="bibr" rid="B4">4</xref>). These can interfere with the everyday life of affected individuals, preventing them from regular activities such as work and house chores (<xref ref-type="bibr" rid="B5">5</xref>).</p>
<p>Social media platforms played a fundamental role in the recognition of this condition, arguably considered the first patient-made illness through online campaigns (<xref ref-type="bibr" rid="B6">6</xref>). Reports of prolonged symptoms after infection were shared by patients on social media in the second half of 2020 and the term &#x0201C;Long COVID&#x0201D; was first adopted on Twitter by Elisa Perego through the hashtag &#x0201C;&#x00023;LongCovid&#x0201D;. People who suffered from long-term effects of Covid-19 also referred to themselves as &#x0201C;long-haulers&#x0201D; (<xref ref-type="bibr" rid="B7">7</xref>).</p>
<p>Several support communities created by people suffering from different health conditions and diseases can be found online (<xref ref-type="bibr" rid="B8">8</xref>). Long-haulers built multiple ones on social media, including Reddit, to share their experiences with the illness, to understand their own condition, to find potential treatments and to provide support for one another (<xref ref-type="bibr" rid="B5">5</xref>).</p>
<p>The perception of Long COVID from a patient&#x00027;s position can differ from what is seen by non-affected individuals. Long-haulers especially requested further efforts from official health organizations, professionals and researchers to acknowledge the reality of their situation and to invest in research and treatment (<xref ref-type="bibr" rid="B9">9</xref>). In addition, patients tend to under-report their negative feelings and symptoms during appointments with healthcare professionals. This social desirability bias complicates the identification of Long COVID profiles (<xref ref-type="bibr" rid="B10">10</xref>).</p>
<p>In this context, conducting patient-centric research that focuses on people with Long COVID and what they share on social media platforms is essential. In fact, social media has been increasingly used for various health research purposes since 2009 (<xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B12">12</xref>). It is a valuable and accessible source of information based on patient-reported outcomes, where most narratives around this new condition emerged and continue to grow. They also centralize the concerns and experiences of affected and neglected individuals on a global scale.</p>
<p>The use of data science and artificial intelligence (AI) methods contributed significantly to COVID-19 research (<xref ref-type="bibr" rid="B13">13</xref>). Similarly, using AI techniques to explore what people with PASC post online would help to uncover patterns and gain a deeper understanding of their unique symptomatic experiences associated with the condition.</p>
<p>Foufi et al. used text mining to extract and identify relationships between biomedical entities from self-reported outcomes shared by people with chronic diseases on Reddit (<xref ref-type="bibr" rid="B14">14</xref>). They emphasized the willingness of affected individuals to share their personal experiences with their illnesses online. This willingness is not limited to those with chronic illnesses as shown by Park and Conway&#x00027;s lexicon-based and supervised machine learning approaches, which confirmed the prevalence of communicative diseases discussions and public health concerns on Reddit (<xref ref-type="bibr" rid="B15">15</xref>). Long COVID was explored by Sarker and Ge through the Reddit community &#x0201C;covidlonghaulers&#x0201D; (<xref ref-type="bibr" rid="B16">16</xref>). Outside of social media, Long COVID research on health records was conducted. Wang et al. employed language processing tools to build a symptoms lexicon for Long COVID from clinical records (<xref ref-type="bibr" rid="B17">17</xref>) while Zhang et al. studied the characteristics of the condition and identified potential subphenotypes using machine learning (<xref ref-type="bibr" rid="B18">18</xref>).</p>
<p>Within this framework, this research work focused on analyzing self-reported Long COVID experiences on Reddit to identify potential profiles of the condition, not only in terms of individual symptoms but also the categories of symptoms and the eventual associations between them. We combined data analysis, lexicon-based and machine learning approaches to process text collected on Long COVID-related Reddit forums, then automatically extracted patterns relevant to the PASC condition.</p>
</sec>
<sec sec-type="materials and methods" id="s2">
<title>Materials and methods</title>
<p>In this section, we develop the different steps to collect, process and transform the data. The implementation of the various phases of the project can be found on this Github repository: HaninAyadi/DOVA_LUX (<ext-link ext-link-type="uri" xlink:href="https://github.com">github.com</ext-link>).</p>
<p><xref ref-type="fig" rid="F1">Figure 1</xref> shows the Reddit data preparation stages before applying the clustering algorithm, which will be detailed in the following sections.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Data collection and preprocessing pipeline. Description of the performed steps for the collection and preprocessing of the data.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpubh-11-1227807-g0001.tif"/>
</fig>
<sec>
<title>Data collection</title>
<p>Reddit was chosen for the data collection for several reasons. First, data is publicly available, easy and free to collect using the Pushshift Application Programming Interface (API). This API was designed to provide enhanced functionality and search capabilities to search through Reddit comments and submissions, which facilitated the task of data collection (<xref ref-type="bibr" rid="B19">19</xref>). Second, Reddit allows posting long content which serves for a better elaboration of the patient experience. Third, during the data collection process, there is no limit to how far back in time you can go. Moreover, it is possible to collect the entire data from the forums at once without needing to set up continuous data collection processes.</p>
<p>The public Reddit forums or communities, also called subreddits, allow pseudonymous Reddit users to share posts, discuss and stay updated on a specific topic of interest while respecting community rules that are supervised by moderators. The existence of those dedicated monitored subreddits reduces the irrelevant content during data collection. The Long COVID subreddits we used are displayed in Supplementary material 1 with their respective number of collected posts and estimated number of members.</p>
<p>The final collection includes all posts since the creation of the forums along with descriptive metadata for each post, such as its id, title, creation timestamp, the pseudonym of the author and information about the subreddit (e.g., name, subscribers). The comments on each post were not collected.</p>
</sec>
<sec>
<title>Data analysis</title>
<p>Most of the posts were originally published in the &#x0201C;covidlonghaulers&#x0201D; subreddit, being the most popular Long COVID community. The raw dataset encompasses 27,216 posts from 8,434 distinct users with an average number of 3 posts per user (SD = 6, Q1 and Q2 = 1, Q3 = 3). The weekly evolution of the general number of posts over the span of 2 years can be seen in <xref ref-type="fig" rid="F2">Figure 2</xref>. We inspected the length of the posts and found that the average number of words per post rounds up to 137 words (SD = 191, Q1 = 38, Q2 = 84, Q3 = 165). The longest post contained 5469 words while the shortest was only one word (which was then excluded during the preprocessing phase).</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Weekly evolution of the number of posts.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpubh-11-1227807-g0002.tif"/>
</fig>
<p>In the subsequent stages of this study, the statistical examination of symptoms, clusters and their associations involves the utilization of quantifiable measures such as percentages, frequencies and co-occurrence analysis.</p>
</sec>
<sec>
<title>Personal/non-personal Long COVID text classifier</title>
<p>It is common that text discussing Long COVID can also include information about the acute infection with COVID-19 as well as news, scientific articles, or anything not descriptive of the personal experience itself with the long-term condition. As our goal was to investigate the experiences and perspectives of people with PASC, we needed to remove portions of the text that were not related to these aspects. For this particular purpose, we built a classifier that identifies sentences with explicit mentions of personally having or suspecting to have Long COVID or being close to someone who does, e.g., a family member.</p>
<p>Two authors (HA, CB) manually labeled a dataset containing a total of 1,395 text samples. This dataset consisted of random sentences from posts. In all, 402 were annotated as related to personal PASC experiences. Complicated cases were discussed between the two authors to obtain a 100% inter-rater agreement.</p>
<p>We then performed several preprocessing steps such as normalization to lowercase, expanding contractions and processing emojis and emoticons on the labeled samples. The 1,395 text examples were split into 1,116 training samples and 279 validation samples. We used BERTweet (<xref ref-type="bibr" rid="B20">20</xref>), a pre-trained language model based on RoBERTa (<xref ref-type="bibr" rid="B21">21</xref>), which stands for Robustly Optimized BERT Pre-training Approach. Its objective is to optimize the training of the BERT (<xref ref-type="bibr" rid="B22">22</xref>) architecture to reduce model pre-training time. Supplementary material 2 shows the classifier performance metrics.</p>
</sec>
<sec>
<title>Data preprocessing</title>
<p>Multiple preprocessing steps were applied to the collected 27,216 English posts. First, we computed the pairwise cosine similarity between the Term Frequency-Inverse Document Frequency (TF-IDF) vectors of the Reddit posts to detect content-based duplicates. Second, the Personal/Non-personal Long COVID text classifier filtered out irrelevant posts and sentences. In fact, for each post, the sentences including an experience with Long COVID were kept. We ended up identifying 20,377 relevant posts. Third, we moved on to a standard preprocessing pipeline of replacing context-specific words and contractions, tokenization, removing punctuation, preprocessing emojis and emoticons, removal of non-ASCII characters, lowercasing and lemmatization.</p>
<p>Stop words are commonly used words such as the words &#x0201C;the&#x0201D; or &#x0201C;and&#x0201D; in English. To fit our context, we manually created a list of context-specific stop words. Amongst the 500 words with the lowest inverse document frequency score in our corpus, we selected those that were unimportant for identifying specific Long COVID patient profiles. Inverse document frequency measures how common a word is in a set of text samples. It is mathematically defined as the logarithm of the quotient: the total number of samples divided by the number of samples where the term is used. The output posts of the preprocessing pipeline are transformed into vectors using the TF-IDF weights.</p>
</sec>
<sec>
<title>K-means clustering</title>
<p>K-means is an unsupervised clustering algorithm. A known challenge with K-means is the choice of K, the initial value of the number of clusters. We experimented with values ranging from 2 to 200 to then closely focus on a smaller interval and evaluate the results for each value. Several criteria were taken into consideration to validate the final clustering result. First, the silhouette score was computed based on the average distances between each data sample, the samples of the same cluster and those of the nearest neighboring cluster. This score is commonly used as an evaluation metric to assess how well the text sample fits in its current cluster compared to the other clusters (<xref ref-type="bibr" rid="B23">23</xref>). Second, the distribution of the posts across the clusters was also considered. For instance, obtaining multiple clusters containing a very low number of posts or obtaining a few enormous clusters both indicate that more values of K should be explored. Finally, the number of extracted symptom groups and their epidemiological interpretation also played an important role in finding the best number of clusters.</p>
<p>The Elbow method was used to approach the value of the number of clusters K. To label the clusters, we read at least the 20 closest text posts to each cluster center and looked at the most frequently used words and word pairs in the clusters.</p>
<p>We used the Python Scikit-learn library&#x00027;s implementation of Mini-Batch K-Means clustering which accelerates the learning for large-scale data. It does incremental updates of the cluster centers&#x00027; positions using mini-batches of the samples instead of the whole set (<xref ref-type="bibr" rid="B24">24</xref>).</p>
<p>In addition, we used the t-distributed stochastic neighbor embedding (t-SNE) method for the projection of the high-dimensional TF-IDF input data vectors in a two-dimensional space (<xref ref-type="bibr" rid="B25">25</xref>). This visualization algorithm allows the discovery of the underlying structure in the data while preserving the similarities found between data points before transitioning to the lower dimensions. The visualizations were considered to assess and understand the results of K-means clustering for different values of K.</p>
<p>Finally, grouping the resulting clusters helped to facilitate the visualization and the assessment of the clustering. The labels used for the grouped symptom clusters referred to the predominant symptom category for each group.</p>
</sec>
<sec>
<title>Symptoms lexicon</title>
<p>A symptom can be referred to by the posters using synonyms, acronyms, or different expressions. Thus, simply recognizing symptoms from the most frequently used words in the posts or each cluster does not allow an accurate identification of symptoms. Wang et al. (<xref ref-type="bibr" rid="B17">17</xref>) developed &#x0201C;PASClex&#x0201D;, a PASC symptom lexicon derived from 328,879 health record clinical notes of 26,117 COVID-19 positive patients in their post-acute infection period. It contains a total of 16,466 synonyms and expressions mapped to 355 different symptoms that are associated with Long COVID. We searched those words in our text data to establish a more accurate occurrence count for each symptom.</p>
</sec>
<sec>
<title>Categorization of the symptoms</title>
<p>We mapped the 355 &#x0201C;PASClex&#x0201D; symptoms to 13 different categories that we identified according to the affected organs and functions. These categories can be seen in Supplementary material 3.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec>
<title>Symptoms extraction</title>
<p>Out of all the processed posts, 78.7% mentioned at least one symptom and 58.42% of all the preprocessed posts contained at least 2 symptoms. The 30 most frequent symptoms across our processed data can be seen in <xref ref-type="fig" rid="F3">Figure 3</xref>. The occurrence reflects the number of posts where the symptom was mentioned at least once. Fatigue, pain, clouded consciousness, anxiety and headaches were all found in more than 12.5% of the posts. We then plotted the co-occurrences of symptoms in <xref ref-type="fig" rid="F4">Figure 4A</xref>. Two symptoms are considered to be co-occurring when they were both mentioned in the same post. For instance, out of the total mentions of depression, around 42% occurred with fatigue, 35% with clouded consciousness and 32% with anxiety. Naturally, the co-occurrence percentage of all symptoms is higher with the most frequent symptoms which explains the darker hues on the left side of the heatmap that become lighter as the symptoms become less frequent. The relationship between the symptoms can be explored further in the network in <xref ref-type="fig" rid="F4">Figure 4B</xref>. The node size reflects the frequency, the directed edges echo the co-occurrence percentage and the color represents the category of the symptoms. There are multiple connections between the different nodes reflecting the high likelihood of experiencing more than one of the frequent symptoms at once. Strong connections mirrored by thick edges are directed toward recurrent symptoms such as fatigue and pain.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Most frequently extracted Long COVID symptoms. The 30 most frequently mentioned symptoms across the Reddit posts. The occurrence is the number of posts where the symptom was reported.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpubh-11-1227807-g0003.tif"/>
</fig>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p><bold>(A, B)</bold> Long COVID symptoms co-occurrence heatmap and network analysis. The heatmap is a representation of the percentage of the co-occurrence of a symptom (on the y-axis) with another symptom (on the x-axis) out of all its occurrences.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpubh-11-1227807-g0004.tif"/>
</fig>
<p>We used the same visualization methods to explore the co-occurrences from a clearer and a more general perspective using the categories of symptoms. In <xref ref-type="fig" rid="F5">Figure 5</xref>, we observe the prevalence of each category of symptoms through the number of posts they appear in. General symptoms (e.g., fatigue and fever) represent the most frequent category with which the remaining categories have high co-occurrence percentages, according to <xref ref-type="fig" rid="F6">Figure 6</xref>. For example, we can see that sleep disorders are strongly connected to general, neurological and ocular, and mental health, psychological and behavioral issues.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Most frequently extracted categories of Long COVID symptoms. The ranking of the 13 categories of symptoms according to the occurrence of their associated symptoms.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpubh-11-1227807-g0005.tif"/>
</fig>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p><bold>(A, B)</bold> Categories of Long COVID co-occurrence heatmap and network analysis. The heatmap is a representation of the percentage of the co-occurrence of a category of symptoms (on the y-axis) with another category of symptoms (on the x-axis) out of all its occurrences.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpubh-11-1227807-g0006.tif"/>
</fig>
</sec>
<sec>
<title>Symptoms clustering</title>
<p>The initial execution of the K-means algorithm on the preprocessed 20,340 posts from 7,560 distinct users returned K = 53 as the optimal number of clusters. However, the following issues were detected for the resulting clusters: a negative silhouette score of &#x02212;0.00124, 8 clusters out of the 53 only contain a single post that could potentially be included in other clusters, multiple clusters were very similar in terms of the subject. We then explored the clusters for K in (25, 53) and found that for K = 35, we have the highest positive silhouette score of 0.00983 and only one cluster amongst the 35 contained a single outlier post which was ignored. We managed to identify the clusters and their key topics. Twenty-eight clusters mostly focused on a specific symptom, two had mentions of mixed symptoms, three contained symptoms triggered by external factors (e.g., exercising), one cluster was vaccine-related and one cluster grouped recovery and relapse experiences.</p>
<p>More details on separate and group labels for the resulting clusters and the word clouds are respectively in Supplementary material 4, 5.</p>
<p>The biggest cluster was composed of around 28% of the total posts. 60% of the posts in this cluster contained at least one symptom from the PASC lexicon. Many of the posts closest to the cluster center were long descriptions of different Long COVID experiences that mention a variety of symptoms. Fatigue was mentioned in almost 11% of the posts, followed by anxiety (8%) and pain (5%). On a bigger scale, mental health, psychological and behavioral symptoms were the most cited with a percentage of 24% out of all posts and followed by general symptoms (e.g., fatigue, around 21%) then neurological and ocular symptoms (e.g., headaches, 14%).</p>
<p>Three clusters contained posts related to post-exertional malaise, symptoms that would be triggered by physical activities and beverages (e.g., alcohol, coffee). The most frequently mentioned symptoms in these clusters were fatigue (16%), dizziness or vertigo (9.3%), clouded consciousness (6.3%), anxiety (4.5%) and headaches (4.1%).</p>
<p>The grouped clusters were projected using the t-SNE algorithm in <xref ref-type="fig" rid="F7">Figure 7</xref>. Visual distinction between several groups and the clusters within the same group is clear despite the presence of some overlapping data points. We chose to only visualize groups of clusters as projecting all of the individual clusters results in a very busy and indistinguishable visualization. We further analyzed those groups by extracting the most frequent symptoms and categories of symptoms within each one. <xref ref-type="table" rid="T1">Table 1</xref> shows the results of this extraction and we can notice the prevalence of the general symptoms that range between 13 and 34% of all symptoms found in any of the different labeled groups of clusters. Fatigue is the general symptom which is almost always present. Neurological and Ocular issues such as headaches are also important and are often present with other types of symptoms. Reasonably, mental health, psychological and behavioral problems, despite not showing in the clustering result as a separate cluster, are still prevalent alongside the other types of symptoms and reflect the struggle of the patients no matter the physical manifestation of their Long COVID.</p>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>Visual representation of the grouped clusters (t-SNE analysis).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpubh-11-1227807-g0007.tif"/>
</fig>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Distribution of the symptoms and the categories of symptoms in the cluster groups.</p></caption> 
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Cluster label</bold></th>
<th valign="top" align="left"><bold>Posts with symptoms</bold></th>
<th valign="top" align="left" colspan="2"><bold>Top individual symptoms</bold></th>
<th valign="top" align="left" colspan="2"><bold>Top categories of symptoms</bold><sup><bold>&#x0002A;</bold></sup></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="3">Vascular/Lymphatic</td>
<td valign="top" align="left" rowspan="3">68.22%</td>
<td valign="top" align="left">Swelling</td>
<td valign="top" align="left">9.04%</td>
<td valign="top" align="left">General</td>
<td valign="top" align="left">16.68%</td>
</tr>
 <tr>
<td valign="top" align="left">Fatigue</td>
<td valign="top" align="left">8.32%</td>
<td valign="top" align="left">Body pain/Mobility</td>
<td valign="top" align="left">15.08%</td>
</tr>
 <tr>
<td valign="top" align="left">Pain</td>
<td valign="top" align="left">7.89%</td>
<td valign="top" align="left">Neurological/Ocular</td>
<td valign="top" align="left">14.52%</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">Cardiorespiratory</td>
<td valign="top" align="left" rowspan="3">86.55%</td>
<td valign="top" align="left">Palpitations</td>
<td valign="top" align="left">7.61%</td>
<td valign="top" align="left">Cardiorespiratory</td>
<td valign="top" align="left">27.24%</td>
</tr>
 <tr>
<td valign="top" align="left">Chest pain</td>
<td valign="top" align="left">7.19%</td>
<td valign="top" align="left">General</td>
<td valign="top" align="left">16.33%</td>
</tr>
 <tr>
<td valign="top" align="left">Fatigue</td>
<td valign="top" align="left">7.07%</td>
<td valign="top" align="left">Mental health/Psychological/Behavioral</td>
<td valign="top" align="left">14.81%</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">Otorhinolaryngology</td>
<td valign="top" align="left" rowspan="3">83.14%</td>
<td valign="top" align="left">Smell/taste problems</td>
<td valign="top" align="left">8.48%</td>
<td valign="top" align="left">Otorhinolaryngology</td>
<td valign="top" align="left">26.78%</td>
</tr>
 <tr>
<td valign="top" align="left">Pain in throat</td>
<td valign="top" align="left">7.39%</td>
<td valign="top" align="left">General</td>
<td valign="top" align="left">15.64%</td>
</tr>
 <tr>
<td valign="top" align="left">Fatigue</td>
<td valign="top" align="left">7.06%</td>
<td valign="top" align="left">Neurological/Ocular</td>
<td valign="top" align="left">12.78%</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">General</td>
<td valign="top" align="left" rowspan="3">91.70%</td>
<td valign="top" align="left">Fever</td>
<td valign="top" align="left">13.55%</td>
<td valign="top" align="left">General</td>
<td valign="top" align="left">33.76%</td>
</tr>
 <tr>
<td valign="top" align="left">Fatigue</td>
<td valign="top" align="left">8.88%</td>
<td valign="top" align="left">Neurological/Ocular</td>
<td valign="top" align="left">12.83%</td>
</tr>
 <tr>
<td valign="top" align="left">Pain</td>
<td valign="top" align="left">6.04%</td>
<td valign="top" align="left">Body pain/Mobility</td>
<td valign="top" align="left">12.11%</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">Sleep</td>
<td valign="top" align="left" rowspan="3">85.40%</td>
<td valign="top" align="left">Insomnia</td>
<td valign="top" align="left">12.51%</td>
<td valign="top" align="left">Sleep</td>
<td valign="top" align="left">20.89%</td>
</tr>
 <tr>
<td valign="top" align="left">Fatigue</td>
<td valign="top" align="left">11.19%</td>
<td valign="top" align="left">General</td>
<td valign="top" align="left">19.55%</td>
</tr>
 <tr>
<td valign="top" align="left">Anxiety</td>
<td valign="top" align="left">7.27%</td>
<td valign="top" align="left">Mental health/Psychological/Behavioral</td>
<td valign="top" align="left">15.53%</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">Neurological/Ocular</td>
<td valign="top" align="left" rowspan="3">92.69%</td>
<td valign="top" align="left">Clouded consciousness</td>
<td valign="top" align="left">18.73%</td>
<td valign="top" align="left">Neurological/Ocular</td>
<td valign="top" align="left">37.98%</td>
</tr>
 <tr>
<td valign="top" align="left">Headaches</td>
<td valign="top" align="left">9.51%</td>
<td valign="top" align="left">General</td>
<td valign="top" align="left">17.02%</td>
</tr>
 <tr>
<td valign="top" align="left">Fatigue</td>
<td valign="top" align="left">9.45%</td>
<td valign="top" align="left">Mental health/Psychological/Behavioral</td>
<td valign="top" align="left">14.19%</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">Body pain/Mobility</td>
<td valign="top" align="left" rowspan="3">95.01%</td>
<td valign="top" align="left">Pain</td>
<td valign="top" align="left">12.16%</td>
<td valign="top" align="left">Body Pain/Mobility</td>
<td valign="top" align="left">30.1%</td>
</tr>
 <tr>
<td valign="top" align="left">Fatigue</td>
<td valign="top" align="left">5.75%</td>
<td valign="top" align="left">General</td>
<td valign="top" align="left">16.26%</td>
</tr>
 <tr>
<td valign="top" align="left">Headaches</td>
<td valign="top" align="left">3.74%</td>
<td valign="top" align="left">Neurological/Ocular</td>
<td valign="top" align="left">14.04%</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">Gastrointestinal</td>
<td valign="top" align="left" rowspan="3">77.92%</td>
<td valign="top" align="left">Fatigue</td>
<td valign="top" align="left">8.04%</td>
<td valign="top" align="left">General</td>
<td valign="top" align="left">22.79%</td>
</tr>
 <tr>
<td valign="top" align="left">Nausea and/or vomiting</td>
<td valign="top" align="left">5.95%</td>
<td valign="top" align="left">Gastrointestinal</td>
<td valign="top" align="left">18.71%</td>
</tr>
 <tr>
<td valign="top" align="left">Pain</td>
<td valign="top" align="left">4.45%</td>
<td valign="top" align="left">Neurological/Ocular</td>
<td valign="top" align="left">13.44%</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">Cutaneous</td>
<td valign="top" align="left" rowspan="3">91.97%</td>
<td valign="top" align="left">Hives</td>
<td valign="top" align="left">17.22%</td>
<td valign="top" align="left">Cutaneous</td>
<td valign="top" align="left">41.19%</td>
</tr>
 <tr>
<td valign="top" align="left">Itching</td>
<td valign="top" align="left">11.96%</td>
<td valign="top" align="left">General</td>
<td valign="top" align="left">13.73%</td>
</tr>
 <tr>
<td valign="top" align="left">Rash</td>
<td valign="top" align="left">10.04%</td>
<td valign="top" align="left">Mental health/Psychological/Behavioral</td>
<td valign="top" align="left">9.85%</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">Women&#x00027;s health</td>
<td valign="top" align="left" rowspan="3">76.83%</td>
<td valign="top" align="left">Fatigue</td>
<td valign="top" align="left">9.49%</td>
<td valign="top" align="left">General</td>
<td valign="top" align="left">21.94%</td>
</tr>
 <tr>
<td valign="top" align="left">Pain</td>
<td valign="top" align="left">8.1%</td>
<td valign="top" align="left">Body pain/Mobility</td>
<td valign="top" align="left">16.54%</td>
</tr>
<tr>
<td valign="top" align="left">Headaches</td>
<td valign="top" align="left">4.46%</td>
<td valign="top" align="left">Neurological/Ocular</td>
<td valign="top" align="left">14.74%</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><sup>&#x0002A;</sup>Symptoms from the PASC lexicon were classified into 13 categories detailed in Supplementary material 3.</p>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<sec>
<title>Long COVID symptomatology</title>
<p>We managed to study the experience of people with Long COVID from their own perspective using Reddit. Text mining techniques such as keyword extraction and Machine Learning algorithms such as classification and clustering were used to better understand the symptomatology and concerns of people with Long COVID.</p>
<p>When it comes to the most frequent symptoms of Long COVID, although prominent symptoms such as pain, fatigue and mental health problems are among the top experienced symptoms across research works, the results in this work slightly differ from previous studies in terms of numbers. Sarker and Ge study on &#x0201C;covidlonghaulers&#x0201D; subreddit posts used the approximate matching method and found that the most frequently reported symptoms were mental health-related (55.2%), fatigue (51.2%), general ache/pain (48.4%), brain fog/confusion (32.8%), and dyspnea (28.9%) (<xref ref-type="bibr" rid="B16">16</xref>). While establishing a Long COVID lexicon from health records, Wang et al. highlighted the most common symptoms in data from clinical records (<xref ref-type="bibr" rid="B17">17</xref>), which can differ from Reddit data, especially in terms of demographics and language vocabulary. The top symptoms they identified were pain (43.1%), anxiety (25.8%), depression (24.0%), fatigue (23.4%), and joint pain (21.0%). On the other hand, we worked on several Long COVID-related subreddits, and using a lexicon-based keyword extraction, the most common symptoms were fatigue (29.4%), pain (22%), clouded consciousness (19.1%), anxiety (17.7%) and headaches (15.6%). The top categories of symptoms were General (45.5%), Neurological/Ocular (42.9%), Mental Health/Psychological/Behavioral (35.2%), Body Pain/Mobility (35.1%) and Cardiorespiratory (31.2%). The differences with the previously cited work on Reddit (<xref ref-type="bibr" rid="B16">16</xref>) could be explained by the different text analysis methods used for the identification of the symptoms as well as the difference in the data sample size affected by the source subreddits, time period of the collection and the inclusion or exclusion of the comments in the analyzed data. Adding to those differences, our analysis of the data was also expanded by measuring the associations between the symptoms through co-occurrence heatmaps, network analysis and clustering. The clustering also helped automatically match posts that did not mention specific symptoms from the lexicon with others discussing the same topics.</p>
<p>Zhang et al. worked on the identification of PASC topics and subphenotypes using cohorts data from the National Patient-Centered Clinical Research Network (<xref ref-type="bibr" rid="B18">18</xref>). They identified in the first place ten distinct topics describing each a set of co-occurring PASC diagnoses. Understandably, symptoms of the same disease category are often mentioned in the same topic, such as diseases of the digestive system and diseases of the musculoskeletal system and connective tissue. Breathing abnormality and throat/chest pain, abdominal and pelvic pain, headache, malaise and fatigue were mentioned in three or more topics. Our results show that pain in general, fatigue and headaches, three of the most frequent symptoms, also have important co-occurrence scores with various other symptoms. Furthermore, the cited work defines four PASC subphenotypes that could be compared to the co-occurrences between the categories of symptoms found in our data. Notably, the body pain and mobility symptom category is mentioned frequently with the neurological and ocular symptom category (around 40% of the time neurological and ocular symptoms were mentioned, they were associated with body pain and mobility issues) which can potentially fall under Subphenotype 3 (musculoskeletal and nervous).</p>
<p>We were able to display the relationship between the various spontaneously reported symptoms of Long COVID and how a variety of them could be experienced at once. High co-occurrence rates were found between fatigue, pain, clouded consciousness, headaches and anxiety. General and Neurological/Ocular symptoms were found important in the different groups of symptoms in our clustering results. In addition, the data used for this research project covered a two-year period which was enriching in the case of a recent condition such as Long COVID. Important clusters related to vaccines and recovery/relapse experiences were also identified.</p>
<p>As discussed in the previous paragraphs, we found similarities with results in research based on clinical records data. This shows how the shared self-reported outcomes from patients all around the world, who probably feel more at ease expressing themselves anonymously to a community with similar interests (<xref ref-type="bibr" rid="B26">26</xref>, <xref ref-type="bibr" rid="B27">27</xref>), could support clinical and epidemiological studies performed on other populations of a smaller scale or a different demographic. It could also help orient clinical research according to the most frequently discussed issues. Generally speaking, this work also proves the complementary value of social media data in health research.</p>
<p>This study has also several limitations. First, text in social media discourse does not necessarily stick to language rules such as grammar and can be informal with extended use of abbreviations (e.g., using acronyms and internet slang) (<xref ref-type="bibr" rid="B28">28</xref>). It is possible that the applied text normalization steps were not exhaustive and did not take into consideration all of those exceptions. In the same context, the usage of a lexicon extracted from clinical records on our input text cannot guarantee the inclusion of all of the symptom vocabulary used in more than 20,000 Reddit posts. Rarely mentioned symptoms that were not part of the lexicon and that did not appear in the most frequent words, could have been overlooked. Furthermore, the classification of the lexicon symptoms in categories was manually done, which could have led to potential misclassifications. Finally, the choice of the number of clusters K after multiple iterations and the manual labeling process do not completely eliminate the possibility of having outliers and non-completely homogenous clusters when it comes to the topic of interest (<xref ref-type="bibr" rid="B29">29</xref>).</p>
</sec>
<sec>
<title>Perspectives</title>
<p>The data collection could be extended to other social media platforms, especially Twitter, the social media where the concept of Long COVID was first depicted and where the first experiences with the condition were shared. This would provide a more inclusive overview in terms of the number and variety of the experiences of the disease itself and the demographic characteristics of the patients such as gender (<xref ref-type="bibr" rid="B30">30</xref>, <xref ref-type="bibr" rid="B31">31</xref>).</p>
<p>Future analysis of how the patient-reported symptoms evolve over time will improve our understanding of the manifestation of Long COVID. Other important concerns of the community such as relapse, recovery, the ability to return to work, the age groups of patients and vaccination are also important to explore in follow-up research works.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="s5">
<title>Conclusion</title>
<p>We have shown that social media is a valuable source of information that helps to orient research in public and digital health. This study gives a thorough analysis of Long COVID symptomatology based on the patient-reported outcomes on Reddit. General symptoms, mainly fatigue, were found to be the most prominent since 2020. Co-occurrence scores between the different categories of symptoms and our network analysis proved that people with Long COVID can experience a wide range of symptoms simultaneously and that the symptom profiles of patients are heterogeneous. Other concerns such as vaccination and relapse after recovery were also put into perspective.</p>
</sec>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The data analyzed in this study is subject to the following licenses/restrictions: The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation. Requests to access these datasets should be directed to <email>guy.fagherazzi&#x00040;lih.lu</email>.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>GF takes full responsibility for the work as a whole, for the decision to submit, publish the manuscript, and designed the research. HA, CB, and GF conducted the research and drafted the article. HA and CB collected, labeled, analyzed, and interpreted the data. AF and MG revised the manuscript critically. All authors contributed to the article and approved the submitted version.</p>
</sec>
</body>
<back>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>This work was supported by the Luxembourg National Research Fund (FNR) (DOVA-LUX project, grant number 16487884) and the Luxembourg Institute of Health (LIH).</p>
</sec>
<ack><p>The authors would like to thank the Luxembourg National Research Fund (FNR) and the Luxembourg Institute of Health (LIH) for their support.</p>
</ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Author disclaimer</title>
<p>The content of this publication is solely the responsibility of the authors and does not necessarily represent the official views of the funders.</p>
</sec>
<sec sec-type="supplementary-material" id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpubh.2023.1227807/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpubh.2023.1227807/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Table_1.XLSX" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Material 1</label>
<caption><p>Members and posts count across subreddits.</p></caption> </supplementary-material>
<supplementary-material xlink:href="Table_2.XLSX" id="SM2" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Material 2</label>
<caption><p>Personal Long COVID text classifier performance.</p></caption> </supplementary-material>
<supplementary-material xlink:href="Table_3.XLSX" id="SM3" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Material 3</label>
<caption><p>Categorization of PASC lexicon symptoms.</p></caption> </supplementary-material>
<supplementary-material xlink:href="Table_4.XLSX" id="SM4" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Material 4</label>
<caption><p>Clusters description.</p></caption> </supplementary-material>
<supplementary-material xlink:href="Image_1.PDF" id="SM5" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Material 5</label>
<caption><p>Word cloud representation of the clusters.</p></caption> </supplementary-material>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>WHO. Coronavirus (COVID-19) Dashboard.</collab></person-group> (<year>2022</year>). Available online at: <ext-link ext-link-type="uri" xlink:href="https://covid19.who.int/">https://covid19.who.int/</ext-link> (accessed November 14, 2022).</citation>
</ref>
<ref id="B2">
<label>2.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>WHO</collab></person-group>. <source>Coronavirus Disease (COVID-19): Post COVID-19 Condition. (March 28, 2023)</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.who.int/news-room/questions-and-answers/item/coronavirus-disease-(covid-19)-post-covid-19-condition">https://www.who.int/news-room/questions-and-answers/item/coronavirus-disease-(covid-19)-post-covid-19-condition</ext-link> (accessed August 9, 2023).</citation>
</ref>
<ref id="B3">
<label>3.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>The White House</collab></person-group>. <article-title>Press Briefing by White House COVID-19 Response Team and Public Health Officials</article-title>. <source>The White House</source>. (<year>2021</year>). Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.whitehouse.gov/briefing-room/press-briefings/2021/02/24/press-briefing-by-white-house-covid-19-response-team-and-public-health-officials-7/">https://www.whitehouse.gov/briefing-room/press-briefings/2021/02/24/press-briefing-by-white-house-covid-19-response-team-and-public-health-officials-7/</ext-link> (accessed November 14, 2022).</citation>
</ref>
<ref id="B4">
<label>4.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>A A Clinical Case Definition of post COVID-19 Condition by a Delphi Consensus 6 October, 2021,.</collab></person-group> (<year>2021</year>). Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.who.int/publications/i/item/WHO-2019-nCoV-Post_COVID-19_condition-Clinical_case_definition-2021.1">https://www.who.int/publications/i/item/WHO-2019-nCoV-Post_COVID-19_condition-Clinical_case_definition-2021.1</ext-link> (accessed November 14, 2022).</citation>
</ref>
<ref id="B5">
<label>5.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rushforth</surname> <given-names>A</given-names></name> <name><surname>Ladds</surname> <given-names>E</given-names></name> <name><surname>Wieringa</surname> <given-names>S</given-names></name> <name><surname>Taylor</surname> <given-names>S</given-names></name> <name><surname>Husain</surname> <given-names>L</given-names></name> <name><surname>Greenhalgh</surname> <given-names>T</given-names></name></person-group>. <article-title>Long COVID&#x02013;the illness narratives</article-title>. <source>Soc Sci Med.</source> (<year>2021</year>) <volume>286</volume>:<fpage>114326</fpage>. <pub-id pub-id-type="doi">10.1016/j.socscimed.2021.114326</pub-id><pub-id pub-id-type="pmid">34425522</pub-id></citation></ref>
<ref id="B6">
<label>6.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Callard</surname> <given-names>F</given-names></name> <name><surname>Perego</surname> <given-names>E</given-names></name></person-group>. <article-title>How and why patients made Long COVID</article-title>. <source>Soc Sci Med.</source> (<year>2021</year>) <volume>268</volume>:<fpage>113426</fpage>. <pub-id pub-id-type="doi">10.1016/j.socscimed.2020.113426</pub-id><pub-id pub-id-type="pmid">33199035</pub-id></citation></ref>
<ref id="B7">
<label>7.</label>
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Yong</surname> <given-names>E</given-names></name></person-group>. <source>COVID-19 Can Last for Several Months. The Atlantic</source>. (<year>2020</year>). Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.theatlantic.com/health/archive/2020/06/covid-19-coronavirus-longterm-symptoms-months/612679/">https://www.theatlantic.com/health/archive/2020/06/covid-19-coronavirus-longterm-symptoms-months/612679/</ext-link> (accessed November 14, 2022).</citation>
</ref>
<ref id="B8">
<label>8.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Conrad</surname> <given-names>P</given-names></name> <name><surname>Bandini</surname> <given-names>J</given-names></name> <name><surname>Vasquez</surname> <given-names>A</given-names></name></person-group>. <article-title>Illness and the internet: from private to public experience</article-title>. <source>Health.</source> (<year>2016</year>) <volume>20</volume>:<fpage>22</fpage>&#x02013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1177/1363459315611941</pub-id><pub-id pub-id-type="pmid">26525400</pub-id></citation></ref>
<ref id="B9">
<label>9.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Berenguera</surname> <given-names>A</given-names></name> <name><surname>Jacques-Avi&#x000F1;&#x000F3;</surname> <given-names>C</given-names></name> <name><surname>Medina-Perucha</surname> <given-names>L</given-names></name> <name><surname>Puente</surname> <given-names>D</given-names></name></person-group>. <article-title>Long term consequences of COVID-19</article-title>. <source>Eur J Intern Med.</source> (<year>2021</year>) <volume>92</volume>:<fpage>34</fpage>&#x02013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1016/j.ejim.2021.08.022</pub-id><pub-id pub-id-type="pmid">34509350</pub-id></citation></ref>
<ref id="B10">
<label>10.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bispo</surname> <given-names>J&#x000FA;nior JP</given-names></name></person-group>. <article-title>Social desirability bias in qualitative health research</article-title>. <source>Rev Saude Publica.</source> (<year>2022</year>) <volume>56</volume>:<fpage>101</fpage>. <pub-id pub-id-type="doi">10.11606/s1518-8787.2022056004164</pub-id><pub-id pub-id-type="pmid">36515303</pub-id></citation></ref>
<ref id="B11">
<label>11.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bour</surname> <given-names>C</given-names></name> <name><surname>Ahne</surname> <given-names>A</given-names></name> <name><surname>Schmitz</surname> <given-names>S</given-names></name> <name><surname>Perchoux</surname> <given-names>C</given-names></name> <name><surname>Dessenne</surname> <given-names>C</given-names></name> <name><surname>Fagherazzi</surname> <given-names>G</given-names></name></person-group>. <article-title>The use of social media for health research purposes: scoping review</article-title>. <source>J Med Internet Res.</source> (<year>2021</year>) <volume>23</volume>:<fpage>e25736</fpage>. <pub-id pub-id-type="doi">10.2196/25736</pub-id><pub-id pub-id-type="pmid">35089149</pub-id></citation></ref>
<ref id="B12">
<label>12.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>J</given-names></name> <name><surname>Wang</surname> <given-names>Y</given-names></name></person-group>. <article-title>Social media use for health purposes: systematic review</article-title>. <source>J Med Internet Res.</source> (<year>2021</year>) <volume>23</volume>:<fpage>e17917</fpage>. <pub-id pub-id-type="doi">10.2196/17917</pub-id><pub-id pub-id-type="pmid">33978589</pub-id></citation></ref>
<ref id="B13">
<label>13.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Saxena</surname> <given-names>N</given-names></name> <name><surname>Gupta</surname> <given-names>P</given-names></name> <name><surname>Raman</surname> <given-names>R</given-names></name> <name><surname>Rathore</surname> <given-names>AS</given-names></name></person-group>. <article-title>Role of data science in managing COVID-19 pandemic</article-title>. <source>Indian Chem Eng.</source> (<year>2020</year>) <volume>62</volume>:<fpage>385</fpage>&#x02013;<lpage>95</lpage>. <pub-id pub-id-type="doi">10.1080/00194506.2020.1855085</pub-id></citation>
</ref>
<ref id="B14">
<label>14.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Foufi</surname> <given-names>V</given-names></name> <name><surname>Timakum</surname> <given-names>T</given-names></name> <name><surname>Gaudet-Blavignac</surname> <given-names>C</given-names></name> <name><surname>Lovis</surname> <given-names>C</given-names></name> <name><surname>Song</surname> <given-names>M</given-names></name></person-group>. <article-title>Mining of textual health information from reddit: analysis of chronic diseases with extracted entities and their relations</article-title>. <source>J Med Internet Res.</source> (<year>2019</year>) <volume>21</volume>:<fpage>e12876</fpage>. <pub-id pub-id-type="doi">10.2196/12876</pub-id><pub-id pub-id-type="pmid">31199327</pub-id></citation></ref>
<ref id="B15">
<label>15.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Park</surname> <given-names>A</given-names></name> <name><surname>Conway</surname> <given-names>M</given-names></name></person-group>. <article-title>Tracking health related discussions on reddit for public health applications</article-title>. <source>AMIA Annu Symp Proc.</source> (<year>2018</year>) <volume>2017</volume>:<fpage>1362</fpage>&#x02013;<lpage>71</lpage>.<pub-id pub-id-type="pmid">29854205</pub-id></citation></ref>
<ref id="B16">
<label>16.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sarker</surname> <given-names>A</given-names></name> <name><surname>Ge</surname> <given-names>Y</given-names></name></person-group>. <article-title>Mining long-COVID symptoms from Reddit: characterizing post-COVID syndrome from patient reports</article-title>. <source>JAMIA Open.</source> (<year>2021</year>) <volume>4</volume>:<fpage>ooab075</fpage>. <pub-id pub-id-type="doi">10.1093/jamiaopen/ooab075</pub-id><pub-id pub-id-type="pmid">34485849</pub-id></citation></ref>
<ref id="B17">
<label>17.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>L</given-names></name> <name><surname>Foer</surname> <given-names>D</given-names></name> <name><surname>MacPhaul</surname> <given-names>E</given-names></name> <name><surname>Lo</surname> <given-names>Y-C</given-names></name> <name><surname>Bates</surname> <given-names>DW</given-names></name> <name><surname>Zhou</surname> <given-names>L</given-names></name></person-group>. <article-title>PASCLex: A comprehensive post-acute sequelae of COVID-19 (PASC) symptom lexicon derived from electronic health record clinical notes</article-title>. <source>J Biomed Inform.</source> (<year>2022</year>) <volume>125</volume>:<fpage>103951</fpage>. <pub-id pub-id-type="doi">10.1016/j.jbi.2021.103951</pub-id><pub-id pub-id-type="pmid">34785382</pub-id></citation></ref>
<ref id="B18">
<label>18.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>H</given-names></name> <name><surname>Zang</surname> <given-names>C</given-names></name> <name><surname>Xu</surname> <given-names>Z</given-names></name> <name><surname>Zhang</surname> <given-names>Y</given-names></name> <name><surname>Xu</surname> <given-names>J</given-names></name> <name><surname>Bian</surname> <given-names>J</given-names></name></person-group>. <article-title>Data-driven identification of post-acute SARS-CoV-2 infection subphenotypes</article-title>. <source>Nat Med.</source> (<year>2022</year>). <volume>5</volume>:<fpage>1</fpage>&#x02013;<lpage>4</lpage>. <pub-id pub-id-type="doi">10.1038/s41591-022-02116-3</pub-id><pub-id pub-id-type="pmid">36456834</pub-id></citation></ref>
<ref id="B19">
<label>19.</label>
<citation citation-type="web"><source>Pushshift Reddit API v4,.0 Documentation &#x02014; Pushshift 4.0 Documentation</source>. (<year>2022</year>). Available online at: <ext-link ext-link-type="uri" xlink:href="https://reddit-api.readthedocs.io/en/latest/">https://reddit-api.readthedocs.io/en/latest/</ext-link> (accessed November 25, 2022).</citation>
</ref>
<ref id="B20">
<label>20.</label>
<citation citation-type="web"><source>Vinai/Bertweet-Base&#x000B7; Hugging Face</source> (<year>2023</year>). Available online at: <ext-link ext-link-type="uri" xlink:href="https://huggingface.co/vinai/bertweet-base">https://huggingface.co/vinai/bertweet-base</ext-link> (accessed February 7, 2023).</citation>
</ref>
<ref id="B21">
<label>21.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y</given-names></name> <name><surname>Ott</surname> <given-names>M</given-names></name> <name><surname>Goyal</surname> <given-names>N</given-names></name> <name><surname>Du</surname> <given-names>J</given-names></name> <name><surname>Joshi</surname> <given-names>M</given-names></name> <name><surname>Chen</surname> <given-names>D</given-names></name> <etal/></person-group>. <article-title>Roberta: A robustly optimized bert pretraining approach</article-title>. <source>arXiv preprint</source>. (<year>2019</year>) arXiv:1907.11692.</citation>
</ref>
<ref id="B22">
<label>22.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Devlin</surname> <given-names>J</given-names></name> <name><surname>Chang</surname> <given-names>MW</given-names></name> <name><surname>Lee</surname> <given-names>K</given-names></name> <name><surname>Toutanova</surname> <given-names>K</given-names></name></person-group>. <article-title>Bert: Pre-training of deep bidirectional transformers for language understanding</article-title>. <source>arXiv preprint</source>. (<year>2018</year>) arXiv:1810.04805.</citation>
</ref>
<ref id="B23">
<label>23.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rousseeuw</surname> <given-names>PJ</given-names></name></person-group>. <article-title>Silhouettes: a graphical aid to the interpretation and validation of cluster analysis</article-title>. <source>J Comput Appl Math.</source> (<year>1987</year>) <volume>20</volume>:<fpage>53</fpage>&#x02013;<lpage>65</lpage>. <pub-id pub-id-type="doi">10.1016/0377-0427(87)90125-7</pub-id></citation>
</ref>
<ref id="B24">
<label>24.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>Sklearn.Cluster.MiniBatchKMeans. Scikit-Learn.</collab></person-group> (<year>2022</year>). Available online at: <ext-link ext-link-type="uri" xlink:href="https://scikit-learn.org/stable/modules/generated/sklearn.cluster.MiniBatchKMeans.html">https://scikit-learn.org/stable/modules/generated/sklearn.cluster.MiniBatchKMeans.html</ext-link> (accessed November 15, 2022).</citation>
</ref>
<ref id="B25">
<label>25.</label>
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Gmail</surname> <given-names>L</given-names></name> <name><surname>Hinton</surname> <given-names>G</given-names></name></person-group>. <source>Visualizing Data using t-SNE</source>. (<year>2008</year>). Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.jmlr.org/papers/volume9/vandermaaten08a/vandermaaten08a.pdf?fbcl">https://www.jmlr.org/papers/volume9/vandermaaten08a/vandermaaten08a.pdf?fbcl</ext-link> (accessed April 3, 2023).</citation>
</ref>
<ref id="B26">
<label>26.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Colineau</surname> <given-names>N</given-names></name> <name><surname>Paris</surname> <given-names>C</given-names></name></person-group>. <article-title>Talking about your health to strangers: understanding the use of online social networks by patients</article-title>. <source>New Rev Hypermedia Multimedia.</source> (<year>2010</year>) <volume>16</volume>:<fpage>141</fpage>&#x02013;<lpage>60</lpage>. <pub-id pub-id-type="doi">10.1080/13614568.2010.496131</pub-id></citation>
</ref>
<ref id="B27">
<label>27.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>De Choudhury</surname> <given-names>M</given-names></name> <name><surname>De</surname> <given-names>S</given-names></name></person-group>. <article-title>Mental health discourse on reddit: self-disclosure, social support, and anonymity</article-title>. <source>ICWSM.</source> (<year>2014</year>) <volume>8</volume>:<fpage>71</fpage>&#x02013;<lpage>80</lpage>. <pub-id pub-id-type="doi">10.1609/icwsm.v8i1.14526</pub-id></citation>
</ref>
<ref id="B28">
<label>28.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Clark</surname> <given-names>E</given-names></name> <name><surname>Araki</surname> <given-names>K</given-names></name></person-group>. <article-title>Text normalization in social media: progress, problems and applications for a pre-processing system of casual English</article-title>. <source>Procedia Soc Behav Sci.</source> (<year>2011</year>) <volume>27</volume>:<fpage>2</fpage>&#x02013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1016/j.sbspro.2011.10.577</pub-id></citation>
</ref>
<ref id="B29">
<label>29.</label>
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Shukla</surname> <given-names>S</given-names></name></person-group>. <source>A Review ON K-means DATA Clustering Approach</source>. (<year>2014</year>). Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.semanticscholar.org/paper/9206564d91ef8e5b64c86639308b6780e930e278">https://www.semanticscholar.org/paper/9206564d91ef8e5b64c86639308b6780e930e278</ext-link> (accessed July 29, 2023).</citation>
</ref>
<ref id="B30">
<label>30.</label>
<citation citation-type="web"><source>Global Twitter User Distribution by Gender 2022. Statista</source>. (<year>2022</year>). Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.statista.com/statistics/828092/distribution-of-users-on-twitter-worldwide-gender/">https://www.statista.com/statistics/828092/distribution-of-users-on-twitter-worldwide-gender/</ext-link> (accessed July 11, 2023).</citation>
</ref>
<ref id="B31">
<label>31.</label>
<citation citation-type="web"><source>Global Reddit User Distribution by Gender 2022. Statista</source>. (<year>2023</year>). Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.statista.com/statistics/1255182/distribution-of-users-on-reddit-worldwide-gender/">https://www.statista.com/statistics/1255182/distribution-of-users-on-reddit-worldwide-gender/</ext-link> (accessed July 11, 2023).</citation>
</ref>
</ref-list> 
</back>
</article> 