<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Big Data</journal-id>
<journal-title>Frontiers in Big Data</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Big Data</abbrev-journal-title>
<issn pub-type="epub">2624-909X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fdata.2025.1615788</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Big Data</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Analyzing student mental health with RoBERTa-Large: a sentiment analysis and data analytics approach</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Khan</surname> <given-names>Hikmat Ullah</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3012994/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Naz</surname> <given-names>Anam</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3148162/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Alarfaj</surname> <given-names>Fawaz Khaled</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x0002A;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Almusallam</surname> <given-names>Naif</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Information Technology, University of Sargodha</institution>, <addr-line>Sargodha</addr-line>, <country>Pakistan</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Management Information Systems, School of Business, King Faisal University</institution>, <addr-line>Al Ahsa</addr-line>, <country>Saudi Arabia</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/152842/overview">Rashid Ibrahim Mehmood</ext-link>, Islamic University of Madinah, Saudi Arabia</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3135894/overview">Daqing Chen</ext-link>, London South Bank University, United Kingdom</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3145648/overview">M. Seenivasan</ext-link>, Annamalai University, India</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Hikmat Ullah Khan <email>dr.hikmat.niazi&#x00040;gmail.com</email></corresp>
<corresp id="c002">Fawaz Khaled Alarfaj <email>falarfaj&#x00040;kfu.edu.sa</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>17</day>
<month>10</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>8</volume>
<elocation-id>1615788</elocation-id>
<history>
<date date-type="received">
<day>21</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>23</day>
<month>09</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Khan, Naz, Alarfaj and Almusallam.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Khan, Naz, Alarfaj and Almusallam</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>The mental health of students plays an important role in their overall wellbeing and academic performance. Growing pressure from academics, co-curricular activities such as sports and personal challenges highlight the need for modern methods of monitoring mental health. Traditional approaches, such as self-reported surveys and psychological evaluations, can be time-consuming and subject to bias. With advancement in artificial intelligence (AI), particularly in natural language processing (NLP), sentiment analysis has emerged as an effective technique for identifying mental health patterns in textual data. However, analyzing students&#x00027; mental health remains a challenging task due to the intensity of emotional expressions, linguistic variations, and context-dependent sentiments. In this study, our primary objective was to investigate the mental health of students by conducting sentiment analysis using advanced deep learning models. To accomplish this task, state-of-the-art Large Language Model (LLM) approaches, such as RoBERTa (a robustly optimized BERT approach), RoBERTa-Large, and ELECTRA, were used for empirical analysis. RoBERTa-Large, an expanded architecture derived from Google&#x00027;s BERT, captures complex patterns and performs more effectively on various NLP tasks. Among the applied algorithms, RoBERTa-Large achieved the highest accuracy of 97%, while ELECTRA yielded 91% accuracy on a multi-classification task with seven diverse mental health status labels. These results demonstrate the potential of LLM-based approaches for predicting students&#x00027; mental health, particularly in relation to the effects of academic and physical activities.</p></abstract>
<kwd-group>
<kwd>large language model</kwd>
<kwd>mental health</kwd>
<kwd>academic performance</kwd>
<kwd>natural language processing</kwd>
<kwd>sentiment analysis</kwd>
</kwd-group>
<counts>
<fig-count count="16"/>
<table-count count="6"/>
<equation-count count="18"/>
<ref-count count="38"/>
<page-count count="19"/>
<word-count count="10554"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Machine Learning and Artificial Intelligence</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Advancements in AI have significantly improved the ability to process vast volumes of textual data, enabling the interpretation of user interactions to extract meaningful insights (<xref ref-type="bibr" rid="B12">Chancellor and De Choudhury, 2020</xref>). One of the most impactful applications of AI is in sentiment analysis, where NLP techniques are used to assess emotions, opinions, and mental states (<xref ref-type="bibr" rid="B19">Ishfaq et al., 2025</xref>). With the rise of social media and digital platforms, individuals frequently express their thoughts, feelings, and experiences through comments, posts, and reviews (<xref ref-type="bibr" rid="B30">Uban et al., 2021</xref>). This user-generated content (UGC) serves as a valuable resource for understanding public sentiment regarding mental health (<xref ref-type="bibr" rid="B25">Primack et al., 2018</xref>). Mental health is a crucial aspect of overall wellbeing, significantly influencing an individual&#x00027;s emotional stability, productivity, and quality of life (<xref ref-type="bibr" rid="B1">Ahmad et al., 2023</xref>). However, analyzing mental health trends based on online sentiment is a challenging task due to the complexity of human emotions, context-dependent language, and diverse expressions of psychological distress (<xref ref-type="bibr" rid="B6">Arag&#x000F3;n et al., 2023</xref>). Social media often reflects a wide spectrum of sentiments, ranging from positive encouragement to severe distress signals related to depression, anxiety, and suicidal thoughts (<xref ref-type="bibr" rid="B26">Roemmich and Andalibi, 2021</xref>). Detecting such emotions accurately requires sophisticated AI-driven models capable of identifying nuanced linguistic patterns and contextual meanings (<xref ref-type="bibr" rid="B4">Alsini et al., 2024</xref>; <xref ref-type="bibr" rid="B23">Li et al., 2025</xref>).</p>
<p>The increasing rate of mental health disorders highlights the importance of AI-driven sentiment analysis. According to studies, the cases of depression and anxiety have significantly increased in recent years, with statistics in the post-pandemic scenario showing a rise of 25% in mental health issues. As also shown in <xref ref-type="fig" rid="F1">Figure 1</xref>, a distressing trend is the increase in suicides, with &#x0003E;700,000 suicides occurring per year around the world (<xref ref-type="bibr" rid="B28">Saraceno and Caldas De Almeida, 2022</xref>). Sentiment analysis on a large scale of UGC can be used to help researchers identify early warning signs, track mental health trends, and develop targeted interventions for coping with psychological distress efficiently (<xref ref-type="bibr" rid="B16">Ding et al., 2023</xref>). Sentiment classification has advanced to a more sophisticated level through the use of deep learning models, such as BERT, GPT, and transformer-based architectures. These models utilize contextual embedding&#x00027;s and attention to identify complex emotional cues in the text data. Adopting an AI-based strategy may help in early detection systems working in mental health and psychology, therapy, and policymaking. Furthermore, categorizing mental health discussions by risk level can help provide people with timely and tailored support (<xref ref-type="bibr" rid="B7">Babu and Kanaga, 2021</xref>).</p>
<fig position="float" id="F1">
<label>Figure 1</label>
<caption><p>Trend analysis of rise in mental health states.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fdata-08-1615788-g0001.tif">
<alt-text>Bar graph showing trends in mental health issues from 2021 to 2024. Categories include Anxiety, Depression, Drug/Substance Use, and Other Mental Health. Notable increases are in Depression reaching 2.53% in 2022 and Other Mental Health peaking at 2.43% in 2024.</alt-text>
</graphic>
</fig>
<p>Sentiment analysis powered by deep learning offers a promising approach to understanding mental health trends in the digital era. With the advancement of AI, it is now possible to utilize AI to bridge the gap between early diagnosis and intervention in mental health research (<xref ref-type="bibr" rid="B37">Zhu, 2023</xref>). In this study, our primary objective is to predict mental health states using sentiment analysis from UGC. For empirical analysis, a self-prepared dataset has been used. For feature extraction and prediction of mental health status, we utilized state-of-the-art transformer-based and baseline models.</p>
<p>The main contributions of this study to sentiment analysis in mental health using deep learning techniques are as follows:</p>
<list list-type="bullet">
<list-item><p>Application of modern transformer models, namely RoBERTa-Large and ELECTRA, in classifying UGC into various mental health statuses. The results show that RoBERTa-Large achieves 97% accuracy, outperforming ELECTRA with 91% accuracy, which highlights the feasibility of using contextual embedding&#x00027;s in sentiment classification.</p></list-item>
<list-item><p>Explored the effect of working with mental health trends by integrating deep learning with NLP. This research leveraged the analysis of sentiment in social media discussions to gain insights into mental health conditions such as depression, anxiety, and suicidal tendencies, aiming to develop a data-driven understanding of mental wellbeing.</p></list-item>
<list-item><p>Developed a robust framework for AI-based mental health monitoring using sentiment analysis, which can be used in mental health support systems to help people by providing timely recommendations based on the sentiment patterns found in UGC.</p></list-item>
</list>
<p>The remainder of the paper, as outlined in <xref ref-type="fig" rid="F2">Figure 2</xref>, is organized as follows: Section 2 presents a comprehensive analysis of the existing literature, with a focus on deep learning techniques. Section 3 provides the roadmap of this study by discussing the steps of the proposed methodology. Section 4 presents a comprehensive analysis of the results, along with a detailed discussion. Section 5 summarizes the study by presenting conclusions and outlining future directions.</p>
<fig position="float" id="F2">
<label>Figure 2</label>
<caption><p>Organization of paper.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fdata-08-1615788-g0002.tif">
<alt-text>Flowchart depicting paper organization in five steps: Step 1 - Introduction, Step 2 - Existing Studies, Step 3 - Research Methodology, Step 4 - Results and Discussion, Step 5 - Conclusion. Each step includes related tasks like significance, deep learning, data preprocessing, exploratory data analysis, and future work. It also involves abstract and references.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2">
<title>2 Related research</title>
<p>Research over the past few years has demonstrated how deep learning and transformer-based models effectively utilize textual data to predict mental health outcomes. Multiple research projects have applied these sophisticated approaches to sentiment evaluation and mental healthcare forecasts because they show substantial promise for spotting depression, along with anxiety and suicidal risk factors. Within current research on mental health prediction, the main challenges that persist include coping with linguistic diversity in expressions, managing ethical issues, and the need to combine multiple data types. <xref ref-type="table" rid="T1">Table 1</xref> presents an analysis of existing studies for a more comprehensive comparative analysis.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Summary analysis of existing studies.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Sr. No</bold>.</th>
<th valign="top" align="left"><bold>Ref</bold></th>
<th valign="top" align="center"><bold>Year</bold></th>
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="left"><bold>Features</bold></th>
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="center"><bold>Results (%)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B11">Bokolo and Liu (2024)</xref></td>
<td valign="top" align="center">2020</td>
<td valign="top" align="left">LSTM</td>
<td valign="top" align="left">HRV from wearables</td>
<td valign="top" align="left">Wearable devices, HRV</td>
<td valign="top" align="center">83</td>
</tr>
<tr>
<td valign="top" align="left">2</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B38">Zogan et al. (2022)</xref></td>
<td valign="top" align="center">2020</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left">Textual features</td>
<td valign="top" align="left">Twitter, Reddit</td>
<td valign="top" align="center">93</td>
</tr>
<tr>
<td valign="top" align="left">3</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B21">Kim et al. (2020)</xref></td>
<td valign="top" align="center">2020</td>
<td valign="top" align="left">Context-DNN</td>
<td valign="top" align="left">Count vectorization</td>
<td valign="top" align="left">Patients&#x00027; data</td>
<td valign="top" align="center">81</td>
</tr>
<tr>
<td valign="top" align="left">4</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B15">Coutts et al. (2020)</xref></td>
<td valign="top" align="center">2021</td>
<td valign="top" align="left">LSTM</td>
<td valign="top" align="left">Textual features</td>
<td valign="top" align="left">Twitter</td>
<td valign="top" align="center">74</td>
</tr>
<tr>
<td valign="top" align="left">5</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B18">Imel et al. (2024)</xref></td>
<td valign="top" align="center">2021</td>
<td valign="top" align="left">BERT-based model</td>
<td valign="top" align="left">Textual features</td>
<td valign="top" align="left">Twitter</td>
<td valign="top" align="center">86</td>
</tr>
<tr>
<td valign="top" align="left">6</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B9">Ben&#x000ED;tez-Andrades et al. (2022)</xref></td>
<td valign="top" align="center">2021</td>
<td valign="top" align="left">Psych BERT</td>
<td valign="top" align="left">Word embedding&#x00027;s</td>
<td valign="top" align="left">social media text</td>
<td valign="top" align="center">65</td>
</tr>
<tr>
<td valign="top" align="left">7</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B10">Boer et al. (2021)</xref></td>
<td valign="top" align="center">2022</td>
<td valign="top" align="left">Bi-LSTM, BERT</td>
<td valign="top" align="left">TF-IDF and PoS</td>
<td valign="top" align="left">social media</td>
<td valign="top" align="center">89</td>
</tr>
<tr>
<td valign="top" align="left">8</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B5">Ankalu Vuyyuru et al. (2023)</xref></td>
<td valign="top" align="center">2022</td>
<td valign="top" align="left">MLM, BiLSTM</td>
<td valign="top" align="left">Word embedding&#x00027;s</td>
<td valign="top" align="left">social media</td>
<td valign="top" align="center">85</td>
</tr>
<tr>
<td valign="top" align="left">9</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B37">Zhu (2023)</xref></td>
<td valign="top" align="center">2023</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">Textual data</td>
<td valign="top" align="left">Twitter, social media</td>
<td valign="top" align="center">92</td>
</tr>
<tr>
<td valign="top" align="left">10</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B22">Kodati and Tene (2023)</xref></td>
<td valign="top" align="center">2023</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left">CBT feature</td>
<td valign="top" align="left">clinical text data</td>
<td valign="top" align="center">91</td>
</tr>
<tr>
<td valign="top" align="left">11</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B14">Chiong et al. (2021)</xref></td>
<td valign="top" align="center">2023</td>
<td valign="top" align="left">MentalBERT</td>
<td valign="top" align="left">Social media interactions</td>
<td valign="top" align="left">Facebook, Twitter</td>
<td valign="top" align="center">76</td>
</tr>
<tr>
<td valign="top" align="left">12</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B32">Verma et al. (2023)</xref></td>
<td valign="top" align="center">2024</td>
<td valign="top" align="left">DeBERTa,</td>
<td valign="top" align="left">behavioral features</td>
<td valign="top" align="left">Sentiment140</td>
<td valign="top" align="center">89</td>
</tr>
<tr>
<td valign="top" align="left">13</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B31">Vajre et al. (2021)</xref></td>
<td valign="top" align="center">2024</td>
<td valign="top" align="left">RoBERTa, BERT</td>
<td valign="top" align="left">Word embedding&#x00027;s</td>
<td valign="top" align="left">textual data</td>
<td valign="top" align="center">79</td>
</tr></tbody>
</table>
</table-wrap>
<sec>
<title>2.1 Existing studies of deep learning</title>
<p>Systematic sentiment analysis in mental health datasets is crucial for researchers investigating depression and anxiety, utilizing data sourced from both wearable devices and social media platforms. While some studies have employed transformer models for depression detection, their findings remain limited due to the lack of comprehensive evaluations across diverse datasets, which reduces the generalizability of the results. Furthermore, insufficient reporting on false positive and false negative detection methods weakens the robustness of their conclusions (<xref ref-type="bibr" rid="B32">Verma et al., 2023</xref>). Comparative analyses of depression and suicide detection using machine learning and transformer models have also overlooked the effects of dataset bias and class imbalance on model performance. The absence of extensive testing across multiple social media platforms restricts the practical applicability of these approaches, highlighting a significant gap in current research (<xref ref-type="bibr" rid="B11">Bokolo and Liu, 2024</xref>).</p>
<p>The research evaluation was limited by a small data sample and unreliable data quality, which decreased the universal applicability of the results regarding wearable technology-based HRV prediction of mental health and overall wellness. Such omissions regarding participant characteristics, including BMI, result in weak generalizations of the study outcomes (<xref ref-type="bibr" rid="B15">Coutts et al., 2020</xref>). The proposed deep learning method for depression intensity measurement on social media did not address the negative effects of noisy data on model precision. The model displayed limited ability to recognize diverse populations because the dataset was not diverse (<xref ref-type="bibr" rid="B17">Ghosh and Anwar, 2021</xref>). Another study suggested using a hybrid deep learning system to detect depression but failed to examine the security issues related to the vulnerability of social media datasets. The methodology did not account for the fact that language usage varies across social media sites, which created limitations for the model&#x00027;s practical application (<xref ref-type="bibr" rid="B38">Zogan et al., 2022</xref>). A deep learning system to identify mental illness from social media did not account for the broad variability of mental symptoms that influence its precision level. The model considered only text information and did not extract corresponding context information regarding user interaction or media post content (<xref ref-type="bibr" rid="B21">Kim et al., 2020</xref>). For predicting the risk of depression, deep models were used; however, they failed to consider that multivariable regression could miss identifying non-linear associations between variables in the model. The performance of this model could be degraded by the lack of diversity in real-world data (<xref ref-type="bibr" rid="B8">Baek and Chung, 2020</xref>). The study on machine learning and deep learning diagnosis techniques did not include data bias analysis, which could affect the fairness of the model. The lack of clarity in the model process negatively impacted transparency, a crucial aspect of designing mental health applications (<xref ref-type="bibr" rid="B20">Kasanneni et al., 2025</xref>).</p>
<p>Important ethical considerations arise when using personal data in a study on predicting mental health consultations from social media posts. The linguistics-based model may have limitations in precision when handling various populations digitally across different platforms (<xref ref-type="bibr" rid="B27">Saha et al., 2022</xref>). The final part of the work involved sentiment analysis through the integration of the Bi-LSTM and BERT models in depression prediction. However, it did not consider textual elements such as sarcasm and irony. Furthermore, the model lacks the capacity to perceive non-verbal cues and multimedia elements, as it relies solely on textual information (<xref ref-type="bibr" rid="B10">Boer et al., 2021</xref>).</p>
</sec>
<sec>
<title>2.2 Existing studies of transformer-based models</title>
<p>Transformers possess exceptional power and capability in capturing long-range dependencies and contextual relationships within sequential data through their self-attention mechanisms. This makes them particularly effective for sentiment analysis tasks, where understanding complex language patterns and context is crucial for accurately detecting sentiment polarity and intensity. In literature, the research model faced trust issues because it solely used texts without integrating multiple types of evidence, and the authors withdrew their work (<xref ref-type="bibr" rid="B36">Zeberga et al., 2022</xref>). Another study employed transformer-based machine learning for counseling conversation analysis, although it failed to address ethical issues related to the use of therapy data. Prediction accuracy could be improved by incorporating non-verbal cues, as text-based analysis often provides insufficient information (<xref ref-type="bibr" rid="B18">Imel et al., 2024</xref>). The assessment of eating disorder-related tweets using machine learning methods alongside BERT models yielded inadequate results because the system design overlooked the diverse expression methods within the text data. Text-only data lacked contextual information about user engagement and multimedia content, which could be vital for understanding the problem (<xref ref-type="bibr" rid="B9">Ben&#x000ED;tez-Andrades et al., 2022</xref>). A transformer-CNN hybrid model for cognitive behavioral therapy assessment in psychological testing did not address the performance inefficiency and resource requirements associated with combining multiple models. The evaluation failed to explore unstructured behavioral and contextual cues, which could enhance both diagnosis accuracy and treatment efficiency (<xref ref-type="bibr" rid="B5">Ankalu Vuyyuru et al., 2023</xref>). The analysis of machine learning algorithms alongside deep learning for mental health diagnosis from social media platforms failed to resolve data quality and imbalance issues. Using textual data alone prevented healthcare professionals from accessing multimodal features, which could enhance diagnostic efficiency (<xref ref-type="bibr" rid="B14">Chiong et al., 2021</xref>). The social media behavioral analysis system, PsychBERT, did not address potential ethical problems that could arise from using individual data. The text-based analysis approach had a potential drawback because it could not detect important contextual signals that multimedia elements and user activity would provide (<xref ref-type="bibr" rid="B31">Vajre et al., 2021</xref>).</p>
<p>A transformer-based deep learning model examined suicidal emotions on social media yet failed to address subtle or hidden expressions because it reduced the system&#x00027;s accuracy level. The approach used only text as its basis while ignoring essential visual and user-related information cues (<xref ref-type="bibr" rid="B22">Kodati and Tene, 2023</xref>). The combination of explainable AI with machine learning analyzed Reddit wellbeing but failed to consider linguistic differences and posting contexts, making it difficult to achieve accurate results. The sole reliance on text data prevented the system from discovering vital behavioral and multimodal information that could strengthen prediction accuracy (<xref ref-type="bibr" rid="B29">Thushari et al., 2023</xref>). The implementation of RoBERTa-Large and BERT in mental healthcare applications restricted their capacity to handle specialized vocabulary within this domain, thus affecting the system&#x00027;s accuracy. Applications could achieve better performance by incorporating non-verbal cues through text analysis alone (<xref ref-type="bibr" rid="B33">Wu et al., 2024</xref>). XAI transformer-based interpretation methods were developed to understand depressed and suicidal user tendencies, although researchers failed to address the difficult nature of detecting subtle expressions. The analysis of text alongside structured data failed to capture non-verbal signals together with situational context, which reduced the accuracy of the interpretation (<xref ref-type="bibr" rid="B24">Malhotra and Jindal, 2024</xref>).</p>
</sec>
</sec>
<sec id="s3">
<title>3 Proposed methodology</title>
<p>The steps of the research methodology are shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. First, data collection, preprocessing, feature extraction, and model training are performed using state-of-the-art NLP models. The main steps in data preprocessing are text normalization, removal of stop words, lemmatization, and tokenization, which refine the raw textual data. Using transformer-based architectures such as RoBERTa-Large, sentiment classification is performed accurately, capturing signs of emotions and mental health patterns from user-generated content. The proposed framework aims to enhance sentiment detection accuracy by designing a model that leverages contextual embedding&#x00027;s and deep learning-based classification.</p>
<fig position="float" id="F3">
<label>Figure 3</label>
<caption><p>Framework of proposed research methodology.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fdata-08-1615788-g0003.tif">
<alt-text>Flowchart depicting text processing for machine learning. Steps include: 1) Data Acquisition of raw text. 2) Preprocessing like tokenization and lemmatization. 3) Cleaning text data. 4) Data Splitting for training and testing. 5) Feature Extraction using embeddings. 6) Model Training with RoBERTa and comparison to ELECTRA. 7) Evaluation of accuracy, precision, recall, and F1 score. 8) Detection of results. Icons accompany each step.</alt-text>
</graphic>
</fig>
<sec>
<title>3.1 Data preprocessing</title>
<p>In this study, data preprocessing is a crucial step to ensure the accuracy and reliability of sentiment analysis on mental health-related content. The raw text data collected from social media platforms contains a lot of noise, such as irrelevant words, special characters, numbers, and stop words, which can negatively impact the performance of Natural Language Processing (NLP) models (<xref ref-type="bibr" rid="B34">Wu et al., 2023</xref>).</p>
<p>Several preprocessing techniques are applied to the textual data to improve its quality (<xref ref-type="bibr" rid="B2">Ahmed et al., 2023</xref>). Initially, stop words <italic>S &#x003F5;</italic> {<italic>s</italic><sub>1</sub>, <italic>s</italic><sub>2</sub>, ...., <italic>s</italic><sub>3</sub>, words <italic>w</italic><sub><italic>j</italic></sub> such as &#x0201C;the&#x0201D; and &#x0201C;is,&#x0201D; which do not carry any useful information in sentiment classification, are removed using function <italic>f</italic><sub><italic>stop</italic></sub> from dataset <italic>D</italic>, as defined in <xref ref-type="disp-formula" rid="E1">Equation 1</xref>. <xref ref-type="table" rid="T2">Table 2</xref> provides a description of the symbols used in the equations for a deeper understanding.</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M14"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mrow><mml:msubsup><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msubsup></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02223;</mml:mo><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02209;</mml:mo><mml:mi>S</mml:mi></mml:mrow><mml:mo stretchy="false">}</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Moreover, set <italic>C</italic> containing digits and characters <italic>c</italic><sub><italic>k</italic></sub> &#x003F5; <italic>C</italic>, such as punctuation marks or emojis, are also eliminated using function <italic>f</italic><sub><italic>char</italic></sub> using <xref ref-type="disp-formula" rid="E2">Equation 2</xref>, for textual uniformity and minimum variations in the dataset.</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M15"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mrow><mml:msubsup><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mo>&#x02033;</mml:mo></mml:msubsup></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02223;</mml:mo><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:msubsup><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msubsup></mml:mrow><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02209;</mml:mo><mml:mi>C</mml:mi></mml:mrow><mml:mo stretchy="false">}</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>After removing the components that might not be relevant, normalized texts are generated to make the content consistent and minimize the differences among word forms, using function <italic>f</italic><sub><italic>norm</italic></sub>, using <xref ref-type="disp-formula" rid="E3">Equation 3</xref>. This happens by converting all text to lowercase, so that when singling out noise words, they are standardized and independent of case sensitivity.</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M16"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle class="mbox"><mml:mtext>lowercase</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:mo>&#x02200;</mml:mo><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:msubsup><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mo>&#x02033;</mml:mo></mml:msubsup></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>After that, we apply lemmatization using <italic>f</italic><sub><italic>lemma</italic></sub> to lower all words to their base forms to thereby increasing the model&#x00027;s efficiency by limiting redundant word variations. For example, words, such as &#x0201C;running,&#x0201D; &#x0201C;ran&#x0201D; are changed to &#x0201C;run,&#x0201D; where similar semantically words are treated equally, as in <xref ref-type="disp-formula" rid="E4">Equation 4</xref>. Lemmatization preserves the contextual meaning of words while reducing data dimensionality, which aids in better model generalization.</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M17"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mrow><mml:msubsup><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mo>&#x02034;</mml:mo></mml:msubsup></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">{</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msubsup><mml:mo>&#x02223;</mml:mo><mml:msubsup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>m</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:msubsup><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mo>&#x02033;</mml:mo></mml:msubsup></mml:mrow></mml:mrow><mml:mo stretchy="false">}</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Another important step is that the NLP model must initially process the text coverage, which involves tokenization <italic>f</italic><sub><italic>token</italic></sub>, in which the text is previously divided into several words or subwords, allowing the NLP model to understand linguistic patterns, as defined in <xref ref-type="disp-formula" rid="E5">Equation 5</xref>. The model can capture syntactic and semantic relationships by breaking down sentences into meaningful terms.</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M18"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mi>o</mml:mi><mml:mi>k</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mrow><mml:msubsup><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mo>&#x02034;</mml:mo></mml:msubsup></mml:mrow></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">}</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Deep learning architectures, such as transformer-based models, are then used to process the tokens generated after preprocessing. Sentiment analysis becomes more efficient when the data are preprocessed, with the noise first removed, text structures standardized to facilitate further data processing, and only the most valuable features selected for classification. This structured approach enhances model performance in detecting sentiment patterns related to mental health concerns such as anxiety, depression, or emotional distress.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Explanation of symbols used in equations.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Symbol</bold></th>
<th valign="top" align="left"><bold>Description</bold></th>
<th valign="top" align="left"><bold>Symbol</bold></th>
<th valign="top" align="left"><bold>Description</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><inline-formula><mml:math id="M1"><mml:mrow><mml:msubsup><mml:mstyle mathvariant="bold"><mml:mtext>d</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>i</mml:mtext></mml:mstyle><mml:mo>&#x02032;</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="left">Redefined document</td>
<td valign="top" align="left"><italic>V</italic></td>
<td valign="top" align="left">Vocabulary size</td>
</tr>
<tr>
<td valign="top" align="left"><bold>T<sub>i</sub>, P<sub>i</sub>, S<sub>i</sub> &#x02208; &#x0211D;<sup>d</sup></bold></td>
<td valign="top" align="left">Token, position, segment embedding of token <italic>x</italic><sub><italic>I</italic></sub></td>
<td valign="top" align="left">&#x003B3;</td>
<td valign="top" align="left">Residual connection</td>
</tr>
<tr>
<td valign="top" align="left"><inline-formula><mml:math id="M3"><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mo>&#x003B1;</mml:mo></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>T</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mo>&#x003B1;</mml:mo></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mo>&#x003B1;</mml:mo></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>S</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mtext>&#x000A0;</mml:mtext><mml:mstyle mathvariant="bold"><mml:mo>&#x003F5;</mml:mo></mml:mstyle><mml:mtext>&#x000A0;</mml:mtext><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>d</mml:mtext></mml:mstyle></mml:mrow></mml:msup></mml:math></inline-formula></td>
<td valign="top" align="left">Learnable scaling factors are applied to the token, position, and segment embeddings, respectively,</td>
<td valign="top" align="left"><inline-formula><mml:math id="M4"><mml:mrow><mml:mi mathvariant="script">M</mml:mi></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="left">Masked token</td>
</tr>
<tr>
<td valign="top" align="left"><inline-formula><mml:math id="M5"><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x000A0;</mml:mtext><mml:mstyle mathvariant="bold"><mml:mo>&#x003F5;</mml:mo></mml:mstyle><mml:mtext>&#x000A0;</mml:mtext><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>d</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msup><mml:mstyle mathvariant="bold"><mml:mtext>d</mml:mtext></mml:mstyle></mml:mrow></mml:msup></mml:math></inline-formula></td>
<td valign="top" align="left">The weight matrix governing the attention mechanism applied to the previous hidden state <inline-formula><mml:math id="M6"><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula></td>
<td valign="top" align="left"><italic>x</italic><sub><italic>t</italic></sub>, <italic>P</italic><sub><italic>t</italic></sub> <italic>and s</italic><sub><italic>t</italic></sub></td>
<td valign="top" align="left">Token, Position, and Segment embeddings</td>
</tr>
<tr>
<td valign="top" align="left"><inline-formula><mml:math id="M7"><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mstyle mathvariant="bold"><mml:mo>&#x003F5;</mml:mo></mml:mstyle><mml:mtext>&#x000A0;</mml:mtext><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>d</mml:mtext></mml:mstyle></mml:mrow></mml:msup></mml:math></inline-formula></td>
<td valign="top" align="left">Bias term for the attention mechanism</td>
<td valign="top" align="left"><italic>M</italic></td>
<td valign="top" align="left">Mask for autoregressive tasks</td>
</tr>
<tr>
<td valign="top" align="left"><bold>&#x003C3;</bold></td>
<td valign="top" align="left">Sigmoid activation function ensuring bounded attention weights.</td>
<td valign="top" align="left"><italic>Q, K, V</italic></td>
<td valign="top" align="left">Linear transformations of the input</td>
</tr>
<tr>
<td valign="top" align="left"><bold>Q</bold><sub><bold>h</bold></sub>, <bold>K</bold><sub><bold>h</bold></sub>, <bold>V</bold><sub><bold>h</bold></sub>, <break/> <inline-formula><mml:math id="M8"><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>R</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mtext>&#x000A0;</mml:mtext><mml:mstyle mathvariant="bold"><mml:mo>&#x003F5;</mml:mo></mml:mstyle><mml:mtext>&#x000A0;</mml:mtext><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>n</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msup><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>d</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>k</mml:mtext></mml:mstyle></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula></td>
<td valign="top" align="left">Query, key, value, and relative positional embeddings matrix for the <italic>h</italic>&#x02212;<italic>th</italic> attention head, respectively.</td>
<td valign="top" align="left"><italic>h</italic><sub><italic>t</italic></sub></td>
<td valign="top" align="left">Hidden state at position <italic>t</italic>.</td>
</tr>
<tr>
<td valign="top" align="left"><bold>C</bold><sub><bold>h</bold></sub></td>
<td valign="top" align="left">A correction vector is added to the value matrix to refine the output representation</td>
<td valign="top" align="left">&#x003BB;</td>
<td valign="top" align="left">Balances the two losses.</td>
</tr>
<tr>
<td valign="top" align="left"><bold>B</bold><sub><bold>h</bold></sub></td>
<td valign="top" align="left">Learned bias matrix applied to the attention logits</td>
<td valign="top" align="left"><inline-formula><mml:math id="M9"><mml:mrow><mml:mi mathvariant="script">D</mml:mi></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="left">Data distribution</td>
</tr>
<tr>
<td valign="top" align="left"><inline-formula><mml:math id="M10"><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>i</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mstyle mathvariant="bold"><mml:mo>&#x003F5;</mml:mo></mml:mstyle><mml:mtext>&#x000A0;</mml:mtext><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>d</mml:mtext></mml:mstyle></mml:mrow></mml:msup></mml:math></inline-formula></td>
<td valign="top" align="left">Input hidden state vector for token <italic>x</italic><sub><italic>i</italic></sub></td>
<td valign="top" align="left"><inline-formula><mml:math id="M11"><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mi>&#x003F5;</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msup><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi><mml:mi>f</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo>;</mml:mo></mml:math></inline-formula><break/> <inline-formula><mml:math id="M12"><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mi>&#x003F5;</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msubsup><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msubsup><mml:mi>d</mml:mi></mml:mrow></mml:msup><mml:mo>;</mml:mo></mml:math></inline-formula><break/> <inline-formula><mml:math id="M13"><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mi>&#x003F5;</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msup><mml:mi>d</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula></td>
<td valign="top" align="left">Weight matrices for the feed-forward layers with an additional weight matrix quadratic term,</td>
</tr>
<tr>
<td valign="top" align="left"><bold>d</bold><sub><bold>k</bold></sub> <bold>and</bold> <bold>d</bold><sub><bold>v</bold></sub></td>
<td valign="top" align="left">Dimensionality of the key and value vectors</td>
<td/>
<td/>
</tr></tbody>
</table>
</table-wrap>
</sec>
<sec>
<title>3.2 Proposed model RoBERTa-Large</title>
<p>RoBERTa-Large (Robustly Optimized BERT Pretraining Approach) is an improved transformer-based model that enhances the performance of BERT by refining its pretraining strategy. RoBERTa-Large is based on the BERT architecture but does not include the Next Sentence Prediction (NSP) objective. It utilizes larger batch sizes and is trained on more data with dynamic masking, developed by Facebook AI. RoBERTa-Large&#x00027;s performance is rich in a variety of tasks centered on NLP, ranging from sentiment analysis to classification to mental health detection, as it has 24 transformer layers, 16 attention heads, and 355 million parameters (<xref ref-type="bibr" rid="B35">Youngmin et al., 2024</xref>). Due to its ability to leverage fine-grained language representations, it is ideal for analyzing complex textual data (e.g., user-generated content about mental health discussions), as proposed by the model architecture defined in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig position="float" id="F4">
<label>Figure 4</label>
<caption><p>Proposed model architecture.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fdata-08-1615788-g0004.tif">
<alt-text>Diagram illustrating a mental illness detection model using BERT and RoBERTa. Input data undergoes token and position embedding, processed by transformer layers with attention blocks and masking. Evaluation measures include accuracy, precision, recall, and F1-score. Outputs classify conditions like anxiety and depression. Key components are numbered for clarity.</alt-text>
</graphic>
</fig>
<sec>
<title>3.2.1 Input embedding&#x00027;s layers</title>
<p>In RoBERTa-Large, the input embedding is responsible for converting input tokens into high-dimensional, dense vectors. These vectors are formed by combining three components.</p>
<p><italic>Token embeddings:</italic> Each token in the input is mapped to a fixed-size vector, representing a learned embedding of the token in the embedding space, as defined in <xref ref-type="disp-formula" rid="E6">Equation 6</xref>.</p>
<p><italic>Position embeddings:</italic> Since the transformer core design does not capture the position of tokens within the sequence, position embeddings must be added to facilitate the understanding of the token positions within the sequence.</p>
<p><italic>Segment embeddings:</italic> If the task involves a sentence pair (e.g., question answering), segment embeddings help identify the two sentences in the pair separately. These embeddings are then combined and fed into the transformer layers. This transformation enables the model to learn about information such as semantics (the meaning of tokens) as well as syntax (the arrangement of tokens in a specific order).</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M19"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>P</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000B7;</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
<sec>
<title>3.2.2 Multi-head self-attention</title>
<p>RoBERTa-Large, which uses a self-attention mechanism for input embedding&#x00027;s that are projected to several attention heads. There is a different aspect of the relationship between tokens that each head learns. Specifically, it learns a query, a key, and a value for each token. The query of a token is compared to the keys of all other tokens, and attention is computed for each token in the sequence (<xref ref-type="bibr" rid="B13">Chen et al., 2024</xref>). A multi-head attention mechanism allows the model to consider relationships and contextual details in parallel, but not at such a fine-grained level as to gain significant expression. The attention computation results of each head are concatenated and then passed through a final linear layer to obtain the attention output calculated using <xref ref-type="disp-formula" rid="E7">Equation 7</xref>.</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M20"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">softmax</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mo>&#x022A4;</mml:mo></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>B</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
<sec>
<title>3.2.3 Feed forwarded network</title>
<p>The combination of the attention layer and the FFN processes the result from the layer. The FFN has two linear layers and one ReLU in between. This part of the network can be a non-linear transformation, which allows the model to better learn more intricate patterns in the data, as shown in <xref ref-type="disp-formula" rid="E8">Equation 8</xref>. Each token&#x00027;s representation is hence transformed independently by the FFN. This is very important because, by examining the information present in the self-attention layers, the FFN enables RoBERTa-Large to establish more powerful, higher-level non-linear relationships.</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M21"><mml:mtable class="eqnarray" columnalign="right"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">ReLU</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B3;</mml:mi><mml:mo>&#x000B7;</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
<sec>
<title>3.2.4 Residual connection and layer normalization</title>
<p>After each sublayer (attention layer or feed-forward network), RoBERTa-Large makes use of residual connections to facilitate gradient flow during training. Through these connections, the input can be added to the output of the sublayer without passing through the sub-layers. It prevents the network from vanishing gradients, allowing the training of deeper models. Layer normalization is executed after adding the residual connection. This allows training to be stabilized by normalizing the hidden states over the layer&#x00027;s output, as in <xref ref-type="disp-formula" rid="E9">Equation 9</xref>.</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M23"><mml:mtable class="eqnarray" columnalign="right"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mi>r</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">LayerNorm</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mtext class="textrm" mathvariant="normal">ReLU</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000B7;</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
<sec>
<title>3.2.5 Masked language modeling (MLM) objective</title>
<p>In pretraining, some of the input tokens are randomly masked, computed using <xref ref-type="disp-formula" rid="E10">Equation 10</xref>. The task in the model is to predict the masked tokens based on the context provided by the surrounding tokens, thereby learning word relationships within a sentence and their context. A standard cross-entropy loss is minimized between the predicted masked tokens and the actual tokens. However, this helps the model learn a robust representation of language, allowing it to generalize very well to downstream NLP tasks as well.</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M25"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mi>L</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x000A0;</mml:mtext><mml:mo>=</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mo>-</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x0220F;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>&#x003F5;</mml:mi><mml:mrow><mml:mi mathvariant="script">M</mml:mi></mml:mrow></mml:mrow></mml:munder></mml:mstyle><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>.</mml:mo><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:msubsup><mml:mrow><mml:mo>&#x0220F;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msubsup></mml:mstyle><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>.</mml:mo><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
</sec>
<sec>
<title>3.3 Proposed model ELECTRA</title>
<p>Efficiently learning an Encoder that Classifies Token Replacements Accurately (ELECTRA) uses a pre-training approach called Replaced Token Detection (RTD). ELECTRA does not predict masked tokens; instead, it learns to distinguish masked tokens from plausible replacements that it generates using a small generator network, as shown in the working architecture defined in <xref ref-type="fig" rid="F5">Figure 5</xref>. The ELECTRA model consists of two components: Generator <italic>G</italic> and Discriminator <italic>D</italic>. The generator is a small, masked language model (MLM) model that generates the masked tokens of a given input sequence. It produces mismatched versions of input but generates plausible replacements for masked tokens. The generator is trained to replace the tokens with replacements that would be generated by the discriminator, a larger transformer. ELECTRA is more efficient and effective than BERT, as it can evaluate each token in the sequence, not just the masked ones (<xref ref-type="bibr" rid="B3">Ahmed et al., 2024</xref>). Both the generator and the discriminator are trained jointly, but only the discriminator is utilized in downstream tasks, such as sentiment analysis.</p>
<fig position="float" id="F5">
<label>Figure 5</label>
<caption><p>Architecture of ELECTRA model.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fdata-08-1615788-g0005.tif">
<alt-text>A diagram illustrating a machine learning model architecture. It consists of two main sections: Generator (left) and Discriminator (right). Both sections include data inputs like stress, sleep time, skills, and productivity. Inputs are masked or processed through an embedded layer for token, positional, and type embeddings. Each section contains an embedded projector with twelve interconnected transformer blocks. The Generator Predictor analyzes for depression and mental health, while the Discriminator processes outputs in original, replaced, or modified forms. The layout shows the flow from data input to output generator.</alt-text>
</graphic>
</fig>
<sec>
<title>3.3.1 Input embedding layer</title>
<p>The tokens they use can be words, subwords, or any other type of token, and they convert these into continuous vector representations, such as those obtained using pre-trained word embedding&#x00027;s, as shown in <xref ref-type="disp-formula" rid="E11">Equation 11</xref>. Additionally, it modifies the input by appending positional encodings to account for the order of the input sequence.</p>
<disp-formula id="E11"><label>(11)</label><mml:math id="M26"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mi>o</mml:mi><mml:mi>k</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>E</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>E</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>g</mml:mi><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
<sec>
<title>3.3.2 Transformer encoder layers</title>
<p>This consists of a stack of transformer encoder blocks that take an embedded input. The layers in these models utilize self-attention and feedforward networks on top of the sequence, handling multi-head self-attention to capture dependencies between tokens within the sequence, as computed in <xref ref-type="disp-formula" rid="E12">Equation 12</xref>. Based on the refined representation of the input, the encoder outputs the text inputs.</p>
<disp-formula id="E12"><label>(12)</label><mml:math id="M27"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">Attention</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>Q</mml:mi><mml:mo>,</mml:mo><mml:mi>K</mml:mi><mml:mo>,</mml:mo><mml:mi>V</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">softmax</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mi>Q</mml:mi><mml:msup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac><mml:mo>&#x0002B;</mml:mo><mml:mi>M</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mi>V</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
<sec>
<title>3.3.3 Generator output layer</title>
<p>In ELECTRA, the generator is an auxiliary MLM that predicts missing text using <xref ref-type="disp-formula" rid="E13">Equation 13</xref>. It learns to produce plausible candidate tokens for the masked positions in the input sequence, matching the existing tokens.</p>
<disp-formula id="E13"><label>(13)</label><mml:math id="M28"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">softmax</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mtext class="textrm" mathvariant="normal">LayerNorm</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
<sec>
<title>3.3.4 Discriminator output layer</title>
<p>The discriminator is responsible for identifying real tokens (those originating from the original data) and fake tokens (those generated by the generator), as defined in <xref ref-type="disp-formula" rid="E14">Equation 14</xref>. It provides a probabilistic answer for each token, indicating whether it is real or fake.</p>
<disp-formula id="E14"><label>(14)</label><mml:math id="M29"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>D</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mover accent='true'><mml:mi>x</mml:mi><mml:mo>&#x0005E;</mml:mo></mml:mover><mml:mi>t</mml:mi></mml:msub><mml:mo>|</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mtext class="textrm" mathvariant="normal">GELU</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mi>o</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mi>o</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
<sec>
<title>3.3.5 Joint training objective</title>
<p>The two tasks are trained jointly: one for the generator (the MLM task, predicting masked tokens) and the other for the discriminator (classifying tokens as real or fake using RTD losses), as shown in <xref ref-type="disp-formula" rid="E15">Equation 15</xref>. The goal is to maximize the discriminator&#x00027;s input to identify fake tokens while minimizing the error made by the generator in producing realistic tokens.</p>
<disp-formula id="E15"><label>(15)</label><mml:math id="M30"><mml:mtable columnalign='right'><mml:mtr><mml:mtd><mml:msub><mml:mi mathvariant="script">L</mml:mi><mml:mrow><mml:mi>E</mml:mi><mml:mi>L</mml:mi><mml:mi>E</mml:mi><mml:mi>C</mml:mi><mml:mi>T</mml:mi><mml:mi>R</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>x</mml:mi><mml:mo>~</mml:mo><mml:mo>&#x1D53B;</mml:mo></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mstyle displaystyle='true'><mml:munder><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi mathvariant="script">M</mml:mi></mml:mrow></mml:munder><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mi>log</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mi>G</mml:mi></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>&#x0007C;</mml:mo><mml:mover accent='true'><mml:mi>x</mml:mi><mml:mo>&#x002DC;</mml:mo></mml:mover></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mi>&#x003BB;</mml:mi><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>T</mml:mi></mml:munderover><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mi>I</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mover accent='true'><mml:mi>x</mml:mi><mml:mo>&#x0005E;</mml:mo></mml:mover><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:mrow></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mrow><mml:mi>log</mml:mi><mml:mi>D</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mover accent='true'><mml:mi>x</mml:mi><mml:mo>&#x0005E;</mml:mo></mml:mover><mml:mi>t</mml:mi></mml:msub><mml:mo>&#x0007C;</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mi>I</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>&#x02260;</mml:mo><mml:msub><mml:mover accent='true'><mml:mi>x</mml:mi><mml:mo>&#x0005E;</mml:mo></mml:mover><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mi>log</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mi>D</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mover accent='true'><mml:mi>x</mml:mi><mml:mo>&#x0005E;</mml:mo></mml:mover><mml:mi>t</mml:mi></mml:msub><mml:mo>&#x0007C;</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
</sec>
<sec>
<title>3.4 Dataset</title>
<p>For the experiments, the dataset is sourced from the Kaggle website, which contains a collection of mental health-related statements from various datasets, including the 3k Conversations Dataset for Chatbot, Depression Reddit Cleaned, Human Stress Prediction, and others. The dataset content is based on reviews generated by users of online platforms such as Reddit and Twitter but is annotated with one of the seven mental health statuses, including normal, depression, suicidal ideation, anxiety, stress, bipolar, and personality disorder. Each entry consists of a unique identifier, a text statement, and a corresponding mental health label. This dataset provides a substantial amount of data for machine learning models to utilize for sentiment analysis and chatbot development, which can aid in early detection and support for mental health conditions.</p>
</sec>
<sec>
<title>3.5 Baseline models</title>
<p>For the comparative analysis, several state-of-the-art deep learning models are considered for detecting complex patterns.</p>
<sec>
<title>3.5.1 LSTM</title>
<p>The LSTM architecture is simplified by the GRU, which uses fewer gates, making it faster to compute and easier to optimize while still addressing vanishing gradient problems. The LSTM has an input, output, and forget gate that regulate the flow of information in a robust manner, allowing for the learning of long-term dependencies.</p>
</sec>
<sec>
<title>3.5.2 Bi-LSTM</title>
<p>From the context provided, BiLSTM enhances the data processing capability of LSTM by processing data in both forward and backward directions, providing the network with significantly more information to work with. This is particularly suitable for tasks that are highly dependent on context, whether in the past or the future.</p>
</sec>
<sec>
<title>3.5.3 GRU</title>
<p>A Gated Recurrent Unit (GRU) is a variant of the standard LSTM network that has streamlined the mechanism for processing sequential data for NLP tasks. The vanishing gradient problem is addressed by GRUs, which have two key gates: the update gate and the reset gate. These gates enable the model to determine which information should be retained and which should be discarded, thereby allowing it to better capture the dependencies of information across time steps, as observed in text data. That is why GRUs are specifically designed for tasks that rely on understanding context and temporal relationships in text, including language modeling, text generation, and sentiment analysis.</p>
</sec>
</sec>
<sec>
<title>3.6 Implementation tools and utilities</title>
<p>The proposed model was developed and fine-tuned using widely adopted deep learning and NLP libraries, including PyTorch and the Hugging Face Transformers framework. Data handling and preprocessing were performed using Pandas and NumPy, while scikit-learn was utilized for evaluation metrics. Visualization and interpretability were supported through Matplotlib and Seaborn, as summarized in <xref ref-type="table" rid="T3">Table 3</xref>. The training and evaluation were conducted on a GPU-enabled environment using Google Colab to ensure efficient computation.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Explanation of experimental setup and resources.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Category</bold></th>
<th valign="top" align="left"><bold>Libraries</bold></th>
<th valign="top" align="center"><bold>Version</bold></th>
<th valign="top" align="left"><bold>Purpose</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Deep learning framework</td>
<td valign="top" align="left">PyTorch</td>
<td valign="top" align="center">2.0.1</td>
<td valign="top" align="left">Core framework for building, training, and deploying the RoBERTa model.</td>
</tr>
<tr>
<td valign="top" align="left">Transformer models</td>
<td valign="top" align="left">Hugging Face Transformers</td>
<td valign="top" align="center">4.30.2</td>
<td valign="top" align="left">Provides pretrained RoBERTa models, tokenizers, and utilities for fine-tuning.</td>
</tr>
<tr>
<td valign="top" align="left">Data handling</td>
<td valign="top" align="left">Pandas</td>
<td valign="top" align="center">2.0.3</td>
<td valign="top" align="left">For loading, cleaning, and managing textual datasets in tabular format.</td>
</tr>
<tr>
<td valign="top" align="left">Numerical computation</td>
<td valign="top" align="left">NumPy</td>
<td valign="top" align="center">1.24.3</td>
<td valign="top" align="left">Efficient numerical operations, array manipulation, and preprocessing support.</td>
</tr>
<tr>
<td valign="top" align="left">Machine learning utilities</td>
<td valign="top" align="left">Scikit-learn</td>
<td valign="top" align="center">1.3.0</td>
<td valign="top" align="left">Train-test split, metrics (accuracy, precision, recall, and F1 score), and confusion matrix.</td>
</tr>
<tr>
<td valign="top" align="left">Visualization</td>
<td valign="top" align="left">Matplotlib</td>
<td valign="top" align="center">3.7.2</td>
<td valign="top" align="left">For plotting training curves, confusion matrices, error analysis, and other visualizations.</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">Seaborn</td>
<td valign="top" align="center">0.12.2</td>
<td valign="top" align="left">High-level visualization for performance metrics and distributions.</td>
</tr>
<tr>
<td valign="top" align="left">Experiment tracking</td>
<td valign="top" align="left">TensorBoard</td>
<td valign="top" align="center">2.13.0</td>
<td valign="top" align="left">Logging and visualization of training metrics.</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">Weights &#x00026; Biases (wandb)</td>
<td valign="top" align="center">0.15.4</td>
<td valign="top" align="left">Tracking training runs, hyperparameters, and performance comparison.</td>
</tr>
<tr>
<td valign="top" align="left">Text preprocessing</td>
<td valign="top" align="left">NLTK</td>
<td valign="top" align="center">3.8.1</td>
<td valign="top" align="left">Tokenization, stopword removal, and lemmatization</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">spaCy</td>
<td valign="top" align="center">3.5.3</td>
<td valign="top" align="left">Advanced linguistic preprocessing (POS, NER, and dependency parsing).</td>
</tr></tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s4">
<title>4 Results and discussion</title>
<p>The empirical analysis-based results are discussed in this section. First, we discuss the descriptive perspective of the datasets, sharing exploratory data analysis, and then the predictive results using applied deep learning models are discussed.</p>
<sec>
<title>4.1 Descriptive analysis</title>
<p>The dataset is compiled from comments shared by students, focusing on their mental health experiences and the impact of education and physical activities on their wellbeing. Students were encouraged to express their thoughts on how various activities impacted their mental state in terms of stress, anxiety, depression, and overall psychological resilience. The responses reflect a wide range of emotions and sentiments related to student activities, academic pressures, and personal challenges. These textual responses are structured and labeled in the dataset, serving as input for sentiment analysis and mental health assessment. Such data can be used by AI models to identify patterns and trends related to the mental wellbeing of students.</p>
<p>The pie chart in <xref ref-type="fig" rid="F6">Figure 6</xref> provides crucial insights into the dataset&#x00027;s composition regarding various mental health statuses. The dataset comprised seven classes with the following distribution: normal (16,351 samples), depression (15,404 samples), suicidal ideation (10,653 samples), anxiety (3,888 samples), bipolar disorder (2,877 samples), stress (2,669 samples), and personality disorder (1,201 samples). The most frequently occurring labels are &#x0201C;depression and suicidal tendencies,&#x0201D; suggesting that the collected data are more likely to involve these labels.</p>
<fig position="float" id="F6">
<label>Figure 6</label>
<caption><p>Count of mental health status.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fdata-08-1615788-g0006.tif">
<alt-text>Pie chart depicting various mental health conditions with different sections: Anxiety, Bipolar, Depression, Suicidal, Stress, Personality disorder, and Normal. Each section varies in size, indicating the proportion of each condition.</alt-text>
</graphic>
</fig>
<p>Significantly also represented is anxiety about its widespread presence in mental health discussions. On the other hand, possible class imbalance is higher in categories like bipolar disorder and personality disorder, and such classes are rare, whereas the described category is common. The dataset distribution also matches with that of real-life trends, where depression and anxiety are more prevalent on social media, and the prevalence of disorders such as bipolar and personality disorders might be less represented, potentially because of stigma and less self-reporting on public forums. For robust NLP models for sentiment analysis, the distribution of labels must be known. It aids in determining the need for techniques to balance feature engineering methods and evaluation metrics to achieve favorable model performance across all categories of mental health. The skewed distribution also suggests that the model should be evaluated carefully, considering a precision&#x02013;recall tradeoff to ensure that predictions for the minority class are reliable.</p>
<p>The correlation heatmap in <xref ref-type="fig" rid="F7">Figure 7</xref> provides insights into the relationships between the various textual features in the dataset. Statement length (<italic>r</italic> = 0.479) and number of words (<italic>r</italic> = 0.466) exhibit a moderate positive correlation with the status variable, indicating that different mental health conditions may require longer statements. Moreover, we observe that the feature statement length and the number of words clearly correlate (<italic>r</italic> = 0.995), as longer statements obviously contain more words. Average word length, however, has a very weak correlation with all other features, suggesting that there is little difference in word complexity within mental health conditions. This proves that textual attributes can be utilized for sentiment analysis in mental health classification.</p>
<fig position="float" id="F7">
<label>Figure 7</label>
<caption><p>Correlation heatmap information about the relationship between the various textual features.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fdata-08-1615788-g0007.tif">
<alt-text>Correlation matrix heatmap showing relationships between four variables: status, statement_length, num_words, and avg_word_length. Dark blue indicates high correlation, while light yellow signifies low correlation. Notable strong correlation seen between statement_length and num_words.</alt-text>
</graphic>
</fig>
<p>The word clouds in <xref ref-type="fig" rid="F8">Figure 8</xref>, from the provided visualizations, indicate the words most frequently used in relation to mental health statuses. The word clouds provide an overall linguistic representation of the user-generated content&#x00027;s depiction of mental health status.</p>
<fig position="float" id="F8">
<label>Figure 8</label>
<caption><p>Word cloud of most frequent words from each class label. <bold>(a)</bold> Overall, <bold>(b)</bold> personality disorder, <bold>(c)</bold> Anxiety, <bold>(d)</bold> normal status category, <bold>(e)</bold> depression, <bold>(f)</bold> suicidal ideation, <bold>(g)</bold> stress-related word cloud, <bold>(h)</bold> bipolar.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fdata-08-1615788-g0008.tif">
<alt-text>Word clouds depict frequently used words in different contexts: 
a) General statements with prominent words like &#x0201C;feel,&#x0201D; &#x0201C;want,&#x0201D; and &#x0201C;life.&#x0201D; 
b) Personality disorder emphasizing &#x0201C;feel&#x0201D; and &#x0201C;im.&#x0201D; 
c) Anxiety highlighting &#x0201C;anxiety&#x0201D; and &#x0201C;im.&#x0201D; 
d) Normal featuring &#x0201C;want&#x0201D; and &#x0201C;one.&#x0201D; 
e) Depression focusing on &#x0201C;know&#x0201D; and &#x0201C;feel.&#x0201D; 
f) Suicidal stressing &#x0201C;feel&#x0201D; and &#x0201C;know.&#x0201D; 
g) Stress with &#x0201C;stress&#x0201D; and &#x0201C;work.&#x0201D; 
h) Bipolar centering on &#x0201C;feel&#x0201D; and &#x0201C;im.&#x0201D; 
The size of words indicates frequency of use.</alt-text>
</graphic>
</fig>
<p><italic>Overall, in Figure (a)</italic>, they feel, want, know, think, live, and other words that dominate this word cloud represent the entire dataset. These words fit within the sphere of introspective and emotional discussions about the disease. &#x0201C;Help&#x0201D; and &#x0201C;work&#x0201D; are also common words that indicate how external support and occupational stress may play a role in shaping the language of mental health.</p>
<p><italic>Personality Disorder in Figure (b):</italic> Words such as &#x0201C;feel,&#x0201D; &#x0201C;life,&#x0201D; &#x0201C;help,&#x0201D; and &#x0201C;think&#x0201D; are scattered within the personality disorder category. This means that people in this category tend to consider their emotions, relationships, and the self. Help implies that many call for external aid or help.</p>
<p><italic>Anxiety in Figure (c):</italic> The word cloud related to anxiety includes &#x0201C;feel,&#x0201D; &#x0201C;think,&#x0201D; &#x0201C;know,&#x0201D; &#x0201C;life,&#x0201D; and &#x0201C;anxiety.&#x0201D; The overthinking that comes through thinking and knowing suggests that anxiety is prominent here. Through the repetition of feeling, the pronounced repetition of feeling highlights how anxious thinking is associated with emotional distress, implying the need for reassurance or professional assistance.</p>
<p>In this <italic>normal status category, as shown in Figure (d)</italic>, the word cloud displays words such as &#x0201C;feel,&#x0201D; &#x0201C;life,&#x0201D; &#x0201C;think,&#x0201D; and &#x0201C;know,&#x0201D; and does not carry the heavy emotional weight of the other categories. This category comprises statements that, in most cases, are not particularly emotional, yet they still involve a neutral or positive discussion. This implies that their terms are balanced and do not have any significant mental health concerns or conversational patterns.</p>
<p><italic>Depression in Figure (e):</italic> Words such as &#x0201C;feel,&#x0201D; &#x0201C;life,&#x0201D; &#x0201C;want,&#x0201D; &#x0201C;help,&#x0201D; and &#x0201C;think&#x0201D; occur most in the word cloud of the depression-related words. Life and want are two things that indicate existential feelings, such as longing or hopelessness. The prominence placed on help highlights the importance of support systems for individuals who are depressed.</p>
<p><italic>Suicidal ideation in Figure (f):</italic> Words in the suicidal ideation category are prominent, including &#x0201C;want&#x0201D; and &#x0201C;life,&#x0201D; and &#x0201C;know&#x0201D; and &#x0201C;feel,&#x0201D; indicative of distressing thoughts and existential, and feeling words. &#x0201C;End&#x0201D; is also used in these statements, suggesting that there is emotional chaos of the worst kind imaginable. This is a pattern of behavior that requires mental health interventions urgently, by people saying such things. Included in the <italic>stress-</italic>related word cloud <italic>in Figure (g)</italic> are the words &#x0201C;stress,&#x0201D; &#x0201C;work,&#x0201D; &#x0201C;time,&#x0201D; and &#x0201C;life,&#x0201D; which all relate to it. This implies that work and time, if they are dominant, are primary factors leading to stress. Therefore, in addition to this, thinking and knowing indicate cognitive strain, where people think about and understand their stressful situations.</p>
<p><italic>Bipolar in Figure (h):</italic> Among the most prominent words in the bipolar disorder category are &#x0201C;go,&#x0201D; &#x0201C;time,&#x0201D; &#x0201C;work,&#x0201D; &#x0201C;feel,&#x0201D; and &#x0201C;help.&#x0201D; Bipolar mood swings are often signaled by fluctuating energy levels and erratic thought processes, both of which point to &#x0201C;go&#x0201D; and &#x0201C;time.&#x0201D;</p>
<p>This analysis, which utilizes word clouds to explore the frequency of the most common words in a corpus, facilitates an understanding of the linguistic markers associated with different mental health conditions.</p>
</sec>
<sec>
<title>4.2 Proposed model results</title>
<p>The proposed model was applied to the student mental health dataset, yielding reliable detection of key psychological patterns and risk factors. The findings highlight the model&#x00027;s potential in supporting early identification and intervention for students&#x00027; wellbeing.</p>
<sec>
<title>4.2.1 LLM model RoBERTa</title>
<p>The RoBERTa-Large model is characterized by a set of carefully tuned hyperparameters that optimize its performance for natural language processing tasks. <xref ref-type="table" rid="T4">Table 4</xref> summarizes the key hyperparameters of the RoBERTa-Large model. To reduce misclassification errors between similar classes, such as depression and bipolar, a weighted cross-entropy loss was used with class-specific weights obtained based on their occurrence frequencies. This hyperparameter tuning ensured that boxes belonging to minority classes have a larger penalty during training, which improves the model&#x00027;s sensitivity toward bipolar instances without affecting performance on depression. The hyperparameters were tuned using a grid search strategy to identify the optimal configuration and enhance the model&#x00027;s generalizability.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Hyperparameter setting of proposed model.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Parameter</bold></th>
<th valign="top" align="left"><bold>Values</bold></th>
<th valign="top" align="left"><bold>Description</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Layers</td>
<td valign="top" align="left">24</td>
<td valign="top" align="left">The number of transformer encoder layers in the model.</td>
</tr>
<tr>
<td valign="top" align="left">Hidden size</td>
<td valign="top" align="left">1,024</td>
<td valign="top" align="left">Dimensionality of the hidden states and embeddings.</td>
</tr>
<tr>
<td valign="top" align="left">Attention heads</td>
<td valign="top" align="left">16</td>
<td valign="top" align="left">Number of self-attention heads in each multi-head attention layer.</td>
</tr>
<tr>
<td valign="top" align="left">Feed-forward size</td>
<td valign="top" align="left">4,096</td>
<td valign="top" align="left">Dimensionality of the intermediate layer in the position-wise feed-forward network.</td>
</tr>
<tr>
<td valign="top" align="left">Max sequence length</td>
<td valign="top" align="left">512</td>
<td valign="top" align="left">The maximum number of tokens the model can process in a single input sequence.</td>
</tr>
<tr>
<td valign="top" align="left">Vocabulary size</td>
<td valign="top" align="left">265</td>
<td valign="top" align="left">Size of the token vocabulary used by the model.</td>
</tr>
<tr>
<td valign="top" align="left">Dropout</td>
<td valign="top" align="left">0.5</td>
<td valign="top" align="left">The dropout rate is applied to prevent overfitting during training.</td>
</tr>
<tr>
<td valign="top" align="left">Attention dropout</td>
<td valign="top" align="left">0.5</td>
<td valign="top" align="left">Dropout rate applied to the attention weights.</td>
</tr>
<tr>
<td valign="top" align="left">Activation function</td>
<td valign="top" align="left">GELU</td>
<td valign="top" align="left">Activation function used in the feed-forward network (Gaussian Error Linear Unit).</td>
</tr>
<tr>
<td valign="top" align="left">Learning rate</td>
<td valign="top" align="left">3e-5</td>
<td valign="top" align="left">Initial learning rate used during pretraining.</td>
</tr>
<tr>
<td valign="top" align="left">Batch size</td>
<td valign="top" align="left">8,192</td>
<td valign="top" align="left">Batch size used during pretraining.</td>
</tr>
<tr>
<td valign="top" align="left">Weight decay</td>
<td valign="top" align="left">0.001</td>
<td valign="top" align="left">L2 regularization was applied to the model weights.</td>
</tr>
<tr>
<td valign="top" align="left">Warmup steps</td>
<td valign="top" align="left">24,000</td>
<td valign="top" align="left">Number of warm-up steps for learning rate scheduling.</td>
</tr>
<tr>
<td valign="top" align="left">Total steps</td>
<td valign="top" align="left">&#x0007E;500,000</td>
<td valign="top" align="left">Total number of training steps during pretraining.</td>
</tr>
<tr>
<td valign="top" align="left">Log function</td>
<td valign="top" align="left">Weighted cross-entropy</td>
<td valign="top" align="left">Used to penalize misclassification of minority classes more heavily.</td>
</tr>
<tr>
<td valign="top" align="left">Class weights</td>
<td valign="top" align="left">Inverse class frequency</td>
<td valign="top" align="left">Weights are assigned proportionally to the inverse frequency of each class.</td>
</tr>
<tr>
<td valign="top" align="left">Sampler</td>
<td valign="top" align="left">Weighted random sampler</td>
<td valign="top" align="left">Ensure balanced mini-batches by oversampling minority classes.</td>
</tr>
<tr>
<td valign="top" align="left">Adam epsilon</td>
<td valign="top" align="left">1e&#x02212;9</td>
<td valign="top" align="left">Term added to the denominator for numerical stability in the Adam optimizer.</td>
</tr>
<tr>
<td valign="top" align="left">Adam beta1</td>
<td valign="top" align="left">0.57</td>
<td valign="top" align="left">Exponential decay rate for the first moment estimates in the Adam optimizer.</td>
</tr>
<tr>
<td valign="top" align="left">Adam beta2</td>
<td valign="top" align="left">0.98</td>
<td valign="top" align="left">Exponential decay rate for the second moment estimates in the Adam optimizer.</td>
</tr>
<tr>
<td valign="top" align="left">Masking probability</td>
<td valign="top" align="left">20%</td>
<td valign="top" align="left">Percentage of tokens masked during the masked language modeling (MLM) objective.</td>
</tr>
<tr>
<td valign="top" align="left">Gradient clipping</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">Maximum gradient norm for gradient clipping to prevent exploding gradients.</td>
</tr></tbody>
</table>
</table-wrap>
<p>Regarding the monitoring and evaluation of mental health in students, leveraging an intensive model (such as RoBERTa-Large-LARGE) in intelligent artificial systems is a new and transformative approach. The deployment of this model in detecting mental conditions using sentiment analysis provides valuable insights into its capabilities and potential applications. The model has an accuracy of 97%, a precision of 95%, a recall of 91%, and an F1 score of 94%. Together, these metrics provide a representation of a superlative model that can reliably and precisely predict the classification of mental health conditions, relying on sentiment data information. This precision is so high as to minimize the risks of false positives, which is especially important when dealing with mental health assessments so that students are not put under unnecessary stress, as shown in the comprehensive analysis of the confusion matrix in <xref ref-type="fig" rid="F9">Figure 9</xref>.</p>
<fig position="float" id="F9">
<label>Figure 9</label>
<caption><p>Confusion matrix of proposed model.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fdata-08-1615788-g0009.tif">
<alt-text>Confusion matrix showing prediction results. Diagonal elements represent correctly predicted instances for categories: Normal (1600), Anxiety (367), Depression (1471), Suicidal (1044), Stress (180), Bipolar (265), Personality disorder (97). Off-diagonal elements represent misclassifications. Color intensity indicates the count, with a scale ranging from 0 to 1600.</alt-text>
</graphic>
</fig>
<p>Similarly, the high recall rate highlights the fact that models can accurately identify true cases of mental health problems with a low likelihood of false negatives. Analysis of the confusion matrix shows that there are significant true positives in terms of diagnosing conditions such as anxiety, bipolar disorder, and depression, with large numbers on the matrix&#x00027;s diagonal. The analysis of training and validation further represents a path of convergence whereby accuracy reaches the upper thresholds, proving that the model is effective at learning. The loss graph in <xref ref-type="fig" rid="F10">Figure 10</xref>, however, is random mainly in the validation loss, which indicates some drops in the model&#x00027;s performance on the validation data at some point(s). Similarly, these spikes are an important indicator of how responsive the model is to specific features or data anomalies, and they aid in further optimization.</p>
<fig position="float" id="F10">
<label>Figure 10</label>
<caption><p>Model performance analysis of loss and accuracy graphs.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fdata-08-1615788-g0010.tif">
<alt-text>Two line graphs depict machine learning metrics over 100 epochs. The left graph shows accuracy (green) and validation accuracy (purple), with both lines fluctuating but generally above 0.5, peaking near 1.0. The right graph shows loss (green) and validation loss (purple), with both lines peaking sharply at several points but generally decreasing.</alt-text>
</graphic>
</fig>
<p>Collectively, the results from model performance demonstrate that these are robust and adaptable AI models. Based on sentiment analysis, the RoBERTa-LARGE model has been shown to accurately and efficiently identify mental health conditions among students, indicating its potential for use in real-world applications. However, the insights suggest that the model must be further refined and recalibrated continuously to achieve higher precision and generalization on other datasets in order to be effective in different environments.</p>
</sec>
<sec>
<title>4.2.2 ELECTRA model</title>
<p>The baseline results of the ELECTRA model in predicting the mental health of students via sentiment analysis prove to be quite strong, with all values of overall accuracy, precision, recall, and F1 score equaling 91%. The model&#x00027;s proficiency in accurately classifying the sentiments in categories such as anxiety, bipolar, depression, normal, personality disorder, stress, and suicidal ideation. Regarding all metrics, this high level of performance indicates that the ELECTRA model effectively interprets small-scale language to identify various mental health conditions, making it an appropriate instrument for early detection and monitoring in academic environments. A detailed view of the model&#x00027;s performance for each class is provided in the confusion matrix. For example, numbers such as &#x0201C;normal&#x0201D; and &#x0201C;suicidal,&#x0201D; which are indicated by the large numbers on the diagonal, indicate a high degree of accuracy of the model in predicting these two states with very few misclassifications, as shown in <xref ref-type="fig" rid="F11">Figure 11</xref>. However, there are unresolved questions as well, for example, between &#x0201C;bipolar&#x0201D; and &#x0201C;depression&#x0201D; or &#x0201C;anxiety&#x0201D; and &#x0201C;stress,&#x0201D; that are highly similar in a clinical or linguistic sense, and which increase the misclassification rates. As a result, this implies that the model, in general, is effective, although it may need to be refined or provided with more training data to better differentiate similar conditions.</p>
<fig position="float" id="F11">
<label>Figure 11</label>
<caption><p>Confusion matrix of ELECTRA model.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fdata-08-1615788-g0011.tif">
<alt-text>Confusion matrix showing predicted versus true labels for mental health conditions. Categories include Normal, Anxiety, Depression, Suicidal, Stress, Bipolar, and Personality Disorder. Most predictions align with true labels, particularly along the diagonal, indicating high accuracy. Color intensity represents the number of correct predictions, with darker colors indicating higher values.</alt-text>
</graphic>
</fig>
<p>The model&#x00027;s learning progress through epochs is visualized by its accuracy and loss graphs, as shown in <xref ref-type="fig" rid="F12">Figure 12</xref>. The accuracy graph remains stable as it converges to high accuracy with the training data, exhibiting minimal overfitting. It is a positive indicator because the validation accuracy closely tracks with the training accuracy, and the model appears to be generalized to new data. The loss graph indicates that the loss is in a downward trend, particularly for the validation loss, with sharp declines following the initial fluctuations. The increase in accuracy confirms this reduction in loss, indicating that the model was reducing error over time. The plot clearly demonstrates the good generalization property that the training and validation performance of the ELECTRA model are well-matched. With that, the model performs well both in learning and operating on new, unseen data it faces&#x02014;a very desirable feature for practical use in many possibly disparate learning environments. With these small variations in accuracy and loss during validation, there is one area where the model can be improved to make it more robust for real-world data variation through regularization and hyperparameter tuning.</p>
<fig position="float" id="F12">
<label>Figure 12</label>
<caption><p>Model performance analysis of loss and accuracy graphs.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fdata-08-1615788-g0012.tif">
<alt-text>Two line graphs display model performance over 100 epochs. The left graph shows training and validation accuracy, with both stabilizing around 0.9 after initial fluctuations. The right graph shows training and validation loss, with both decreasing sharply initially and then stabilizing, though validation loss shows more fluctuation.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec>
<title>4.3 Comparison of the proposed model with the baseline models</title>
<p>By examining confusion matrices that predict the mental health of students using ELECTRA, GRU, LSTM, and Bi-LSTM models, this study provides a clear understanding of the strengths and limitations of these approaches. These models were subsequently evaluated under various conditions, annotated as anxiety, bipolar, depression, normal, personality disorder, stress, and suicidal ideation, which show the predictive capabilities and inadequacies by means of rates of correct and incorrect classification for the model.</p>
<sec>
<title>4.3.1 GRU model</title>
<p>Results showing 77% accuracy and an F1 score are based on the GRU model&#x00027;s predictions, which are not particularly impressive. With the intention of comparing this model to others, the latter has high true positive rates of 3,081 (normal) and 1,371 (suicidal) states in its confusion matrix. Its accuracy is poor, however, when it is trying to differentiate more closely related disorders, such as anxiety vs. bipolar, where it is perhaps less clear where the confusion is coming from, as seen in <xref ref-type="fig" rid="F13">Figure 13</xref>. Since there are misclassification rates in the reactions to classes, the GRU is not very effective at differentiating between the types of nuanced emotional expressions. This suggests that, although the GRU architecture is computationally fast and effective for simple categories of conditions, it may not be the most successful design for tasks that require a deep exploration of fine-grained health conditions.</p>
<fig position="float" id="F13">
<label>Figure 13</label>
<caption><p>Confusion matrix of GRU model.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fdata-08-1615788-g0013.tif">
<alt-text>Heatmap displaying the distribution of various mental health conditions. Axes list conditions: anxiety, bipolar, depression, normal, personality disorder, stress, and suicidal. Darker green indicates higher values. Notable data points include 3081 normals and 1371 suicidals with normals.</alt-text>
</graphic>
</fig>
</sec>
<sec>
<title>4.3.2 LSTM model</title>
<p>It achieves an accuracy score of 79%, which outperforms the F1 score and reflects a deeper ability to model sequence-type data and their longer dependencies in this context. The LSTM confusion matrix shows strong potential for classifying normal and depression states. However, performance is hindered by the significant overlap between symptoms of depression and suicidal ideation&#x02014;for example, 657 depression cases were misclassified as suicidal ideation, as illustrated in <xref ref-type="fig" rid="F14">Figure 14</xref>. This issue highlights the challenge for reliable sentiment analysis, as the textual cues for sorrowful feelings and suicidal ideation are often closely related. Nevertheless, LSTM remains generally strong in handling a variety of mental health labels.</p>
<fig position="float" id="F14">
<label>Figure 14</label>
<caption><p>Confusion matrix of LSTM model.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fdata-08-1615788-g0014.tif">
<alt-text>Confusion matrix for mental health condition predictions. Rows represent actual conditions and columns represent predicted conditions: Anxiety, Bipolar, Depression, Normal, Personality disorder, Stress, Suicidal. Highlights include high correct predictions for Normal (3044), Depression (2243), and Suicidal (1556). Incorrect predictions are more frequent in lower numbers, indicating some misclassifications between different conditions. A gradient from light to dark green represents frequency, with darker shades indicating higher counts.</alt-text>
</graphic>
</fig>
</sec>
<sec>
<title>4.3.3 Bi-LSTM model</title>
<p>The BiLSTM model exhibits the most robust performance, with an accuracy of 81% and an F1 score of 79%, and is therefore the most accurate among the three. It has a superbly well-distributed accuracy value in its confusion matrix, shown in <xref ref-type="fig" rid="F15">Figure 15</xref>, with particularly good &#x0201C;normal&#x0201D; and &#x0201C;suicidal&#x0201D; state correct classification rates. As a result, the Bi-LSTM can process data from both past and future input sequences bidirectionally, offering a comprehensive understanding of the data. This allows for considerable confusion reduction compared to other models, especially in distinguishing overlapping symptoms in various conditions. Regarding predictive performance and more general network capabilities related to mental health analysis, with higher accuracy and sensitivity, the Bi-LSTM is regarded as the model of choice for mental health monitoring, particularly in critical applications.</p>
<fig position="float" id="F15">
<label>Figure 15</label>
<caption><p>Confusion matrix of Bi-LSTM model.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fdata-08-1615788-g0015.tif">
<alt-text>Confusion matrix heatmap depicting predictions against actual categories such as Anxiety, Bipolar, Depression, Normal, Personality disorder, Stress, and Suicidal. Color intensity varies from light to dark green, indicating frequency from zero to three thousand. Notable values include 3026 (Normal-Normal), 2214 (Depression-Depression), and 1606 (Suicidal-Suicidal).</alt-text>
</graphic>
</fig>
<p>The performance of the RoBERTa-LARGE and ELECTRA models demonstrates their strengths and differences when applied to the sentiment analysis of mental health conditions among students. As shown in <xref ref-type="table" rid="T5">Table 5</xref>, the RoBERTa-LARGE model achieves slightly higher accuracy and is robust enough to handle complex linguistic features and nuances, as evidenced by the significantly higher values of precision and recall compared to the state-of-the-art model. However, the training accuracy and validation accuracy do not align as well, suggesting that the model may be overfitting. In contrast, the ELECTRA model presents an overall score (accuracy, precision, recall, and F1 score) of 91% and less variation between the training and validation metrics (although with lower absolute performance). Finally, both models suggest confusion areas between closely similar mental health conditions such as bipolar and depression, which are a particular challenge to the sentiment analysis when fine emotional expressions need to be interpreted in a sophisticated way. Thus, ELECTRA may be advantageous over RoBERTa-Large in the real world, where generalizability across potentially heterogeneous data is important. Modeling different deep learning models in sentiment analysis for mental health classification, such as GRU, LSTM, BiLSTM, ELECTRA, and RoBERTa-Large, exhibits the ordering of their performance. At baseline performance levels, as measured by the GRU model, the accuracy is 77%. It is still a low percentage, but this is because the model is quite simple and lacks sufficient information to provide a comprehensive result on the emotional state of the situation. With LSTM and BiLSTM, there is a significant improvement in learning dependencies in sequence data, resulting in an F1 score of 79%. The ELECTRA model performs well on many metrics, achieving an overall accuracy of 91%, which is attributed to its transformer-based architecture that pushes contextual learning to the extreme. However, the best RoBERTa-Large model is the clear winner in terms of maximum performance, achieving 97% accuracy and a 94% F1 score, demonstrating its powerful capacity in capturing fine-grained language representations that describe language specificity related to mental health states. The sequence of these two developments recalls the significant impact that the arrival of advanced NLP technologies may have on improving the quality and credibility of mental health assessments based on text.</p>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Comparison of applied models (Results in %).</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="center"><bold>Accuracy</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>F1 score</bold></th>
<th valign="top" align="center"><bold>Macro F1</bold></th>
<th valign="top" align="center"><bold>Micro F1</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">GRU</td>
<td valign="top" align="center">77</td>
<td valign="top" align="center">75</td>
<td valign="top" align="center">74</td>
<td valign="top" align="center">77</td>
<td valign="top" align="center">76</td>
<td valign="top" align="center">77</td>
</tr>
<tr>
<td valign="top" align="left">LSTM</td>
<td valign="top" align="center">79</td>
<td valign="top" align="center">79</td>
<td valign="top" align="center">78</td>
<td valign="top" align="center">79</td>
<td valign="top" align="center">78</td>
<td valign="top" align="center">79</td>
</tr>
<tr>
<td valign="top" align="left">BiLSTM</td>
<td valign="top" align="center">81</td>
<td valign="top" align="center">75</td>
<td valign="top" align="center">78</td>
<td valign="top" align="center">79</td>
<td valign="top" align="center">78</td>
<td valign="top" align="center">81</td>
</tr>
<tr>
<td valign="top" align="left">ELECTRA</td>
<td valign="top" align="center">91</td>
<td valign="top" align="center">91</td>
<td valign="top" align="center">91</td>
<td valign="top" align="center">91</td>
<td valign="top" align="center">91</td>
<td valign="top" align="center">91</td>
</tr>
<tr>
<td valign="top" align="left">RoBERTa-Large</td>
<td valign="top" align="center">97</td>
<td valign="top" align="center">95</td>
<td valign="top" align="center">91</td>
<td valign="top" align="center">94</td>
<td valign="top" align="center">93</td>
<td valign="top" align="center">97</td>
</tr></tbody>
</table>
</table-wrap>
<p>In <xref ref-type="table" rid="T5">Table 5</xref>, the macro- and micro-average F1 scores allow for a relatively more balanced evaluation in the presence of class imbalance. It is true that traditional RNN-based models, such as GRU, LSTM, and BiLSTM, perform well with macro F1 scores ranging from 76 to 78%, but transformer-based approaches greatly improve this performance. ELECTRA is the most consistent across different metrics, but RoBERTa-Large is the best-performing among all baseline models, with a macro F1 score of 93% and a micro F1 score of 97%, indicating its stronger generalization to minority classes while retaining high overall accuracy.</p>
<p>This analysis strongly supports the use of advanced model architectures, such as RoBERTa-Large, for contextually sensitive and intertwined tasks, including analyzing text data to predict patients&#x00027; mental health status, as depicted in <xref ref-type="fig" rid="F16">Figure 16</xref>. These models provide gains in both predictive accuracy and reliability, qualities essential for applications where the precise meaning of emotional and psychological states must be assessed as accurately as possible.</p>
<fig position="float" id="F16">
<label>Figure 16</label>
<caption><p>Applied model performance analysis.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fdata-08-1615788-g0016.tif">
<alt-text>Bar chart comparing different models&#x00027; performances. GRU scores around 75, LSTM slightly higher, BiLSTM near 80, ELECTRA above 90, and Proposed RoBERTa-Large scoring the highest at nearly 95.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec>
<title>4.4 Comparison with existing studies</title>
<p><xref ref-type="table" rid="T6">Table 6</xref> presents various feature-based models applied in mental health cases to yield results. A comparative analysis of the proposed RoBERTa-Large model with existing models shows that our model achieves a 97% result in sentiment analysis of mental health, relying on word embedding&#x00027;s. Indeed, this model surpasses other approaches, such as the traditional LSTM, which utilizes HRV from wearable devices, as well as textual features from Twitter, to achieve results of 83% and 74% in earlier applications. Among more advanced models, such as BiLSTM&#x0002B;BERT and other models like MentalBERT and RoBERTa&#x0002B;BERT, which perform well on complex text data and social media interactions, the highest performance achieved is only up to 89% and 79%, respectively. For example, a more recent development, XLNet, which utilizes word embedding&#x00027;s for social text-based communication, was more effective than RoBERTa-Large, achieving an outcome of 92%. Based on the above, the RoBERTa-Large model is selected for its superior performance, as well as its effective handling of word embedding&#x00027;s and robust training, which likely corresponds to a deeper, more nuanced understanding of language context for mental health. This makes it a brilliant tool in the domain of mental health diagnostics and sentiment analysis.</p>
<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>Comparison with existing studies.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Ref</bold></th>
<th valign="top" align="center"><bold>Year</bold></th>
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="center"><bold>Results (%)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><xref ref-type="bibr" rid="B11">Bokolo and Liu (2024)</xref></td>
<td valign="top" align="center">2020</td>
<td valign="top" align="left">LSTM</td>
<td valign="top" align="left">HRV data</td>
<td valign="top" align="center">83</td>
</tr>
<tr>
<td valign="top" align="left"><xref ref-type="bibr" rid="B15">Coutts et al. (2020)</xref></td>
<td valign="top" align="center">2021</td>
<td valign="top" align="left">LSTM</td>
<td valign="top" align="left">Twitter</td>
<td valign="top" align="center">74</td>
</tr>
<tr>
<td valign="top" align="left"><xref ref-type="bibr" rid="B10">Boer et al. (2021)</xref></td>
<td valign="top" align="center">2022</td>
<td valign="top" align="left">Bi-LSTM</td>
<td valign="top" align="left">Social media</td>
<td valign="top" align="center">89</td>
</tr>
<tr>
<td valign="top" align="left"><xref ref-type="bibr" rid="B14">Chiong et al. (2021)</xref></td>
<td valign="top" align="center">2023</td>
<td valign="top" align="left">MentalBERT</td>
<td valign="top" align="left">Facebook, Twitter</td>
<td valign="top" align="center">76</td>
</tr>
<tr>
<td valign="top" align="left"><xref ref-type="bibr" rid="B31">Vajre et al. (2021)</xref></td>
<td valign="top" align="center">2024</td>
<td valign="top" align="left">RoBERTa,</td>
<td valign="top" align="left">textual data</td>
<td valign="top" align="center">79</td>
</tr>
<tr>
<td valign="top" align="left"><xref ref-type="bibr" rid="B8">Baek and Chung (2020)</xref></td>
<td valign="top" align="center">2025</td>
<td valign="top" align="left">XLNet</td>
<td valign="top" align="left">Social media</td>
<td valign="top" align="center">92</td>
</tr>
<tr>
<td valign="top" align="left" colspan="2">Proposed</td>
<td valign="top" align="left">RoBERTa-Large</td>
<td valign="top" align="left">Sentiment analysis on mental health</td>
<td valign="top" align="center">97</td>
</tr></tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s5">
<title>5 Conclusion and future research</title>
<p>The role of activities in shaping students&#x00027; physical and mental wellbeing is significant, which makes monitoring their mental health essential. Sentiment analysis, facilitated by advancements in AI, provides a powerful means to assess and understand the psychological states of sports students from the vast quantities of textual data they generate. The findings demonstrate that this technology provides valuable insights into the complexities of mental health patterns within this demographic. We have utilized a suite of AI-based models to capture subtle linguistic cues that reflect different mental health concerns. Of these models, our proposed RoBERTa-Large (96.5%) has performed impressively, achieving over 97% accuracy in the task of detecting and interpreting mental health-related sentiments. With its capability to process word embeddings and fine-tune training using big data, this model achieves the highest level of precision among existing models, making it an invaluable aid in addressing the mental health issues of students. This finding demonstrates the success of applying advanced AI models such as RoBERTa-Large in psychological health analysis. Moreover, it highlights the potential of AI models to revolutionize the way we perform analysis and treat people regarding mental health in educational environments. Going forward, there are numerous opportunities to monitor and improve students&#x00027; mental health by utilizing AI-based sentiment analysis. Most importantly, combining other forms of data, including video, audio, and physiological measurements, with text analysis will enable a richer description of the student&#x00027;s mental state. These extra-modality data may represent non-verbal or physiological cues that are unavailable when using text only and could lead to a more precise and thorough assessment, ultimately increasing the reach of AI in mental health evaluations and efforts to incorporate these technologies into daily life, thereby improving support systems for students.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec sec-type="ethics-statement" id="s7">
<title>Ethics statement</title>
<p>Ethical approval was not required for this study, as no human participants were involved. The dataset utilized in the research was downloaded from Kaggle, which is freely available for research purposes, at: <ext-link ext-link-type="uri" xlink:href="https://www.kaggle.com/datasets/kreeshrajani/3k-conversations-dataset-for-chatbot">https://www.kaggle.com/datasets/kreeshrajani/3k-conversations-dataset-for-chatbot</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>HK: Conceptualization, Methodology, Supervision, Writing &#x02013; original draft. AN: Data curation, Formal analysis, Visualization, Writing &#x02013; review &#x00026; editing. FA: Funding acquisition, Resources, Validation, Writing &#x02013; review &#x00026; editing. NA: Funding acquisition, Project administration, Investigation, Writing &#x02013; review &#x00026; editing.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work was supported by the Deanship of Scientific Research, Vice Presidency for Graduate Studies and Scientific Research, King Faisal University, Saudi Arabia [Grant No. KFU253383].</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ahmad</surname> <given-names>W.</given-names></name> <name><surname>Khan</surname> <given-names>H. U.</given-names></name> <name><surname>Iqbal</surname> <given-names>T.</given-names></name> <name><surname>Iqbal</surname> <given-names>S.</given-names></name></person-group> (<year>2023</year>). <article-title>Attention-based multi-channel gated recurrent neural networks: a novel feature-centric approach for aspect-based sentiment classification</article-title>. <source>IEEE Access</source> <volume>11</volume>, <fpage>54408</fpage>&#x02013;<lpage>54427</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2023.3281889</pub-id></citation>
</ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ahmed</surname> <given-names>M.</given-names></name> <name><surname>Khan</surname> <given-names>H. U.</given-names></name> <name><surname>Khan</surname> <given-names>M. A.</given-names></name> <name><surname>Tariq</surname> <given-names>U.</given-names></name> <name><surname>Kadry</surname> <given-names>S.</given-names></name></person-group> (<year>2023</year>). <article-title>Contextaware answer selection in community question answering exploiting spatial temporal bidirectional long short-term memory</article-title>. <source>ACM Trans. Asian Low Resour. Lang. Inf. Process</source> <volume>22</volume>:<fpage>130</fpage>. <pub-id pub-id-type="doi">10.1145/3603398</pub-id></citation>
</ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ahmed</surname> <given-names>M.</given-names></name> <name><surname>Khan</surname> <given-names>H. U.</given-names></name> <name><surname>Munir</surname> <given-names>E. U.</given-names></name></person-group> (<year>2024</year>). <article-title>Conversational AI: an explication of few-shot learning problem in transformers-based Chabot systems</article-title>. <source>IEEE Trans. Comput. Soc. Syst.</source> <volume>11</volume>, <fpage>1888</fpage>&#x02013;<lpage>1906</lpage>. <pub-id pub-id-type="doi">10.1109/TCSS.2023.3281492</pub-id></citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alsini</surname> <given-names>R.</given-names></name> <name><surname>Naz</surname> <given-names>A.</given-names></name> <name><surname>Khan</surname> <given-names>H. U.</given-names></name> <name><surname>Bukhari</surname> <given-names>A.</given-names></name> <name><surname>Daud</surname> <given-names>A.</given-names></name> <name><surname>Ramzan</surname> <given-names>M.</given-names></name></person-group> (<year>2024</year>). <article-title>Using deep learning and word embeddings for predicting human agreeableness behavior</article-title>. <source>Sci. Rep</source>. <volume>14</volume>:<fpage>29875</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-024-81506-8</pub-id><pub-id pub-id-type="pmid">39622946</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ankalu Vuyyuru</surname> <given-names>V.</given-names></name> <name><surname>Vamsi Krishna</surname> <given-names>G.</given-names></name> <name><surname>Christal Mary</surname> <given-names>D.</given-names></name> <name><surname>Mohammed Sulayman Alsubayhay</surname> <given-names>A.</given-names></name> <name><surname>Professor</surname> <given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>A transformer-CNN hybrid model for cognitive behavioral therapy in psychological assessment and intervention for enhanced diagnostic accuracy and treatment efficiency</article-title>. <source>Int. J. Adv. Comput. Sci. Appl.</source> <volume>14</volume>:<fpage>594</fpage>. <pub-id pub-id-type="doi">10.14569/IJACSA.2023.0140766</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Arag&#x000F3;n</surname> <given-names>M. E.</given-names></name> <name><surname>L&#x000F3;pez-Monroy</surname> <given-names>A. P.</given-names></name> <name><surname>Gonz&#x000E1;lez-Gurrola</surname> <given-names>L. C.</given-names></name> <name><surname>Montes-y-G&#x000F3;mez</surname> <given-names>M</given-names></name></person-group>. <article-title>Detecting mental disorders in social media through emotional patterns - the case of anorexia depression</article-title>. (<year>2023</year>). <source>IEEE Trans. Affect Comput.</source> <volume>14</volume>, <fpage>211</fpage>&#x02013;<lpage>222</lpage>. <pub-id pub-id-type="doi">10.1109/TAFFC.2021.3075638</pub-id></citation>
</ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Babu</surname> <given-names>N. V.</given-names></name> <name><surname>Kanaga</surname> <given-names>E. G. M.</given-names></name></person-group> (<year>2021</year>). <article-title>Sentiment analysis in social media data for depression detection using artificial intelligence: a review</article-title>. <source>SN Comput. Sci.</source> <volume>3</volume>:<fpage>74</fpage>. <pub-id pub-id-type="doi">10.1007/s42979-021-00958-1</pub-id><pub-id pub-id-type="pmid">34816124</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Baek</surname> <given-names>J.-W.</given-names></name> <name><surname>Chung</surname> <given-names>K.</given-names></name></person-group> (<year>2020</year>). <article-title>Context deep neural network model for predicting depression risk using multiple regression</article-title>. <source>IEEE Access</source> <volume>8</volume>, <fpage>18171</fpage>&#x02013;<lpage>18181</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2020.2968393</pub-id></citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ben&#x000ED;tez-Andrades</surname> <given-names>J. A.</given-names></name> <name><surname>Alija-P&#x000E9;rez</surname> <given-names>J.-M.</given-names></name> <name><surname>Vidal</surname> <given-names>M.-E.</given-names></name> <name><surname>Pastor-Vargas</surname> <given-names>R.</given-names></name> <name><surname>Garc&#x000ED;a-Ord&#x000E1;s</surname> <given-names>M. T.</given-names></name></person-group> (<year>2022</year>). <article-title>Traditional machine learning models and bidirectional encoder representations from transformer (BERT)&#x02013;based automatic classification of tweets about eating disorders: algorithm development and validation study</article-title>. <source>JMIR Med. Inform</source>. <volume>10</volume>:<fpage>e34492</fpage>. <pub-id pub-id-type="doi">10.2196/34492</pub-id><pub-id pub-id-type="pmid">35200156</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Boer</surname> <given-names>M.</given-names></name> <name><surname>Stevens</surname> <given-names>G. W. J. M.</given-names></name> <name><surname>Finkenauer</surname> <given-names>C.</given-names></name> <name><surname>de Looze</surname> <given-names>M. E.</given-names></name> <name><surname>van den Eijnden</surname> <given-names>R. J. J. M.</given-names></name></person-group> (<year>2021</year>). <article-title>Social media use intensity, social media use problems, and mental health among adolescents: investigating directionality and mediating processes</article-title>. <source>Comput. Human Behav.</source> <volume>116</volume>:<fpage>106645</fpage>. <pub-id pub-id-type="doi">10.1016/j.chb.2020.106645</pub-id></citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bokolo</surname> <given-names>B. G.</given-names></name> <name><surname>Liu</surname> <given-names>Q.</given-names></name></person-group> (<year>2024</year>). <article-title>Advanced comparative analysis of machine learning and transformer models for depression and suicide detection in social media texts</article-title>. <source>Electronics</source> <volume>13</volume>:<fpage>3980</fpage>. <pub-id pub-id-type="doi">10.3390/electronics13203980</pub-id></citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chancellor</surname> <given-names>S.</given-names></name> <name><surname>De Choudhury</surname> <given-names>M.</given-names></name></person-group> (<year>2020</year>). <article-title>Methods in predictive techniques for mental health status on social media: a critical review</article-title>. <source>NPJ Digit Med.</source> <volume>3</volume>:<fpage>43</fpage>. <pub-id pub-id-type="doi">10.1038/s41746-020-0233-7</pub-id><pub-id pub-id-type="pmid">32219184</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>M.</given-names></name> <name><surname>Lu</surname> <given-names>P.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Enhancing Chinese comprehension and reasoning for large language models: an efficient LoRA fine-tuning and tree of thoughts framework</article-title>. <source>J. Supercomput</source>. <volume>81</volume>:<fpage>50</fpage>. <pub-id pub-id-type="doi">10.1007/s11227-024-06499-7</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chiong</surname> <given-names>R.</given-names></name> <name><surname>Budhi</surname> <given-names>G. S.</given-names></name> <name><surname>Dhakal</surname> <given-names>S.</given-names></name> <name><surname>Chiong</surname> <given-names>F.</given-names></name></person-group> (<year>2021</year>). <article-title>A textual-based featuring approach for depression detection using machine learning classifiers and social media texts</article-title>. <source>Comput. Biol. Med.</source> <volume>135</volume>:<fpage>104499</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2021.104499</pub-id><pub-id pub-id-type="pmid">34174760</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Coutts</surname> <given-names>L. V.</given-names></name> <name><surname>Plans</surname> <given-names>D.</given-names></name> <name><surname>Brown</surname> <given-names>A. W.</given-names></name> <name><surname>Collomosse</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>Deep learning with wearable based heart rate variability for prediction of mental and general health</article-title>. <source>J. Biomed. Inform.</source> <volume>112</volume>:<fpage>103610</fpage>. <pub-id pub-id-type="doi">10.1016/j.jbi.2020.103610</pub-id><pub-id pub-id-type="pmid">33137470</pub-id></citation></ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ding</surname> <given-names>J.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name> <name><surname>Lu</surname> <given-names>P.</given-names></name> <name><surname>Yang</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Du</surname> <given-names>Y.</given-names></name></person-group> (<year>2023</year>). <article-title>DialogueINAB: an interaction neural network based on attitudes and behaviors of interlocutors for dialogue emotion recognition</article-title>. <source>J. Supercomput.</source> <volume>79</volume>, <fpage>20481</fpage>&#x02013;<lpage>20514</lpage>. <pub-id pub-id-type="doi">10.1007/s11227-023-05439-1</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ghosh</surname> <given-names>S.</given-names></name> <name><surname>Anwar</surname> <given-names>T.</given-names></name></person-group> (<year>2021</year>). <article-title>Depression intensity estimation via social media: a deep learning approach</article-title>. <source>IEEE Trans. Comput. Soc. Syst.</source> <volume>8</volume>, <fpage>1465</fpage>&#x02013;<lpage>1474</lpage>. <pub-id pub-id-type="doi">10.1109/TCSS.2021.3084154</pub-id></citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Imel</surname> <given-names>Z. E.</given-names></name> <name><surname>Tanana</surname> <given-names>M. J.</given-names></name> <name><surname>Soma</surname> <given-names>C. S.</given-names></name> <name><surname>Hull</surname> <given-names>T. D.</given-names></name> <name><surname>Pace</surname> <given-names>B. T.</given-names></name> <name><surname>Stanco</surname> <given-names>S. C.</given-names></name></person-group> (<year>2024</year>). <article-title>Mental health counseling from conversational content with transformer-based machine learning</article-title>. <source>JAMA Netw. Open</source> <volume>7</volume>:<fpage>E2352590</fpage>. <pub-id pub-id-type="doi">10.1001/jamanetworkopen.2023.52590</pub-id><pub-id pub-id-type="pmid">38252437</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ishfaq</surname> <given-names>U.</given-names></name> <name><surname>Khan</surname> <given-names>H. U.</given-names></name> <name><surname>Shabbir</surname> <given-names>D.</given-names></name></person-group> (<year>2025</year>). <article-title>Exploring the role of sentiment analysis with network and temporal features for finding influential users in social media platforms</article-title>. <source>Soc. Netw. Anal. Min</source>. <volume>14</volume>:<fpage>241</fpage>. <pub-id pub-id-type="doi">10.1007/s13278-024-01396-6</pub-id></citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kasanneni</surname> <given-names>Y.</given-names></name> <name><surname>Duggal</surname> <given-names>A.</given-names></name> <name><surname>Sathyaraj</surname> <given-names>R.</given-names></name> <name><surname>Raja</surname> <given-names>S. P.</given-names></name></person-group> (<year>2025</year>). <article-title>Effective analysis of machine and deep learning methods for diagnosing mental health using social media conversations</article-title>. <source>IEEE Trans. Comput. Soc. Syst</source>. <volume>12</volume>, <fpage>274</fpage>&#x02013;<lpage>294</lpage>. <pub-id pub-id-type="doi">10.1109/TCSS.2024.3487168</pub-id></citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>J.</given-names></name> <name><surname>Lee</surname> <given-names>J.</given-names></name> <name><surname>Park</surname> <given-names>E.</given-names></name> <name><surname>Han</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>A deep learning model for detecting mental illness from user content on social media</article-title>. <source>Sci. Rep</source>. <volume>10</volume>:<fpage>11846</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-020-68764-y</pub-id><pub-id pub-id-type="pmid">32678250</pub-id></citation></ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kodati</surname> <given-names>D.</given-names></name> <name><surname>Tene</surname> <given-names>R.</given-names></name></person-group> (<year>2023</year>). <article-title>Identifying suicidal emotions on social media through transformer-based deep learning</article-title>. <source>Appl. Intell.</source> <volume>53</volume>, <fpage>11885</fpage>&#x02013;<lpage>11917</lpage>. <pub-id pub-id-type="doi">10.1007/s10489-022-04060-8</pub-id></citation>
</ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>D.</given-names></name> <name><surname>Tang</surname> <given-names>N.</given-names></name> <name><surname>Chandler</surname> <given-names>M.</given-names></name> <name><surname>Nanni</surname> <given-names>E.</given-names></name></person-group> (<year>2025</year>). <article-title>An optimal approach for predicting cognitive performance in education based on deep learning</article-title>. <source>Comput. Human Behav</source>. <volume>167</volume>:<fpage>108607</fpage>. <pub-id pub-id-type="doi">10.1016/j.chb.2025.108607</pub-id></citation>
</ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Malhotra</surname> <given-names>A.</given-names></name> <name><surname>Jindal</surname> <given-names>R.</given-names></name></person-group> (<year>2024</year>). <article-title>XAI transformer based approach for interpreting depressed and suicidal user behavior on online social networks</article-title>. <source>Cogn. Syst. Res.</source> <volume>84</volume>:<fpage>101186</fpage>. <pub-id pub-id-type="doi">10.1016/j.cogsys.2023.101186</pub-id></citation>
</ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Primack</surname> <given-names>B. A.</given-names></name> <name><surname>Shensa</surname> <given-names>A.</given-names></name> <name><surname>Sidani</surname> <given-names>J. E.</given-names></name> <name><surname>Bowman</surname> <given-names>N.</given-names></name> <name><surname>Knight</surname> <given-names>J.</given-names></name> <name><surname>Karim</surname> <given-names>S. A.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>&#x0201C;Reducing risk for mental health conditions associated with social media use: encouraging &#x0201C;REAL&#x0201D; communication,&#x0201D;</article-title> in <source>Families and Technology</source>, eds. <person-group person-group-type="editor"><name><surname>Van Hook</surname> <given-names>J.</given-names></name> <name><surname>McHale</surname> <given-names>S. M.</given-names></name> <name><surname>King</surname> <given-names>V.</given-names></name></person-group> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <fpage>155</fpage>&#x02013;<lpage>176</lpage>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Roemmich</surname> <given-names>K.</given-names></name> <name><surname>Andalibi</surname> <given-names>N.</given-names></name></person-group> (<year>2021</year>). <article-title>Data Subjects&#x00027; conceptualizations of and attitudes toward automatic emotion recognition-enabled wellbeing interventions on social media</article-title>. <source>Proc. ACM Hum.-Comput. Interact.</source> <volume>5</volume>:<fpage>CSCW2</fpage>. <pub-id pub-id-type="doi">10.1145/3476049</pub-id></citation>
</ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Saha</surname> <given-names>K.</given-names></name> <name><surname>Yousuf</surname> <given-names>A.</given-names></name> <name><surname>Boyd</surname> <given-names>R. L.</given-names></name> <name><surname>Pennebaker</surname> <given-names>J. W.</given-names></name> <name><surname>De Choudhury</surname> <given-names>M.</given-names></name></person-group> (<year>2022</year>). <article-title>Social media discussions predict mental health consultations on college campuses</article-title>. <source>Sci. Rep.</source> <volume>12</volume>:<fpage>123</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-021-03423-4</pub-id><pub-id pub-id-type="pmid">34996909</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Saraceno</surname> <given-names>B.</given-names></name> <name><surname>Caldas De Almeida</surname> <given-names>J. M.</given-names></name></person-group> (<year>2022</year>). <article-title>An outstanding message of hope: The WHO World Mental Health Report 2022</article-title>. <source>Epidemiol. Psychiatr. Sci.</source> <volume>31</volume>:<fpage>e53</fpage>. <pub-id pub-id-type="doi">10.1017/S2045796022000373</pub-id><pub-id pub-id-type="pmid">35833232</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Thushari</surname> <given-names>P. D.</given-names></name> <name><surname>Aggarwal</surname> <given-names>N.</given-names></name> <name><surname>Vajrobol</surname> <given-names>V.</given-names></name> <name><surname>Saxena</surname> <given-names>G. J.</given-names></name> <name><surname>Singh</surname> <given-names>S.</given-names></name> <name><surname>Pundir</surname> <given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>Identifying discernible indications of psychological well-being using ML: explainable AI in reddit social media interactions</article-title>. <source>Soc. Netw. Anal. Min</source>. <volume>13</volume>:<fpage>141</fpage>. <pub-id pub-id-type="doi">10.1007/s13278-023-01145-1</pub-id></citation>
</ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Uban</surname> <given-names>A.-S.</given-names></name> <name><surname>Chulvi</surname> <given-names>B.</given-names></name> <name><surname>Rosso</surname> <given-names>P.</given-names></name></person-group> (<year>2021</year>). <article-title>An emotion and cognitive based analysis of mental health disorders from social media data</article-title>. <source>Future Gener. Comput. Syst.</source> <volume>124</volume>, <fpage>480</fpage>&#x02013;<lpage>494</lpage>. <pub-id pub-id-type="doi">10.1016/j.future.2021.05.032</pub-id></citation>
</ref>
<ref id="B31">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Vajre</surname> <given-names>V.</given-names></name> <name><surname>Naylor</surname> <given-names>M.</given-names></name> <name><surname>Kamath</surname> <given-names>U.</given-names></name> <name><surname>Shehu</surname> <given-names>A.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;PsychBERT: a mental health language model for social media mental health behavioral analysis,&#x0201D;</article-title> in <source>2021 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</source> (<publisher-loc>Houston, TX</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1077</fpage>&#x02013;<lpage>1082</lpage>.</citation>
</ref>
<ref id="B32">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Verma</surname> <given-names>S.</given-names></name> <name><surname>Vishal</surname> <given-names>Joshi, R. C.</given-names></name> <name><surname>Dutta</surname> <given-names>M. K.</given-names></name> <name><surname>Jezek</surname> <given-names>S.</given-names></name> <name><surname>Burget</surname> <given-names>R.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;AI-enhanced mental health diagnosis: leveraging transformers for early detection of depression tendency in textual data,&#x0201D;</article-title> in <source>2023 15th International Congress on Ultra Modern Telecommunications and Control Systems and Workshops (ICUMT)</source>, (<publisher-loc>Ghent</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>56</fpage>&#x02013;<lpage>61</lpage>.</citation>
</ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>L.</given-names></name> <name><surname>Liu</surname> <given-names>P.</given-names></name> <name><surname>Zhao</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>P.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name></person-group> (<year>2024</year>). <article-title>Human cognition-based consistency inference networks for multi-modal fake news detection</article-title>. <source>IEEE Trans. Knowl. Data Eng.</source> <volume>36</volume>, <fpage>211</fpage>&#x02013;<lpage>225</lpage>. <pub-id pub-id-type="doi">10.1109/TKDE.2023.3280555</pub-id></citation>
</ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>L.</given-names></name> <name><surname>Long</surname> <given-names>Y.</given-names></name> <name><surname>Gao</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name></person-group> (<year>2023</year>). <article-title>MFIR: multimodal fusion and inconsistency reasoning for explainable fake news detection</article-title>. <source>Inf. Fusion</source> <volume>100</volume>:<fpage>101944</fpage>. <pub-id pub-id-type="doi">10.1016/j.inffus.2023.101944</pub-id></citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Youngmin</surname> <given-names>L.</given-names></name> <name><surname>Andrew</surname> <given-names>L. S. I. D.</given-names></name> <name><surname>Duoduo</surname> <given-names>C.</given-names></name> <name><surname>Stephen</surname> <given-names>W. R.</given-names></name></person-group> (<year>2024</year>). <article-title>The role of model architecture and scale in predicting molecular properties: insights from fine-tuning RoBERTa, BART, and LLaMA</article-title>. <source>arXiv</source> [Preprint]. <italic>arXiv:2405.00949</italic>. <pub-id pub-id-type="doi">10.48550/arXiv.2405.00949</pub-id></citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zeberga</surname> <given-names>K.</given-names></name> <name><surname>Attique</surname> <given-names>M.</given-names></name> <name><surname>Shah</surname> <given-names>B.</given-names></name> <name><surname>Ali</surname> <given-names>F.</given-names></name> <name><surname>Jembre</surname> <given-names>Y. Z.</given-names></name> <name><surname>Chung</surname> <given-names>T.-S.</given-names></name></person-group> (<year>2022</year>). <article-title>[Retracted] A novel text mining approach for mental health prediction using Bi-LSTM and BERT model</article-title>. <source>Comput. Intell. Neurosci.</source> <volume>2022</volume>:<fpage>7893775</fpage>. <pub-id pub-id-type="doi">10.1155/2022/7893775</pub-id><pub-id pub-id-type="pmid">35281185</pub-id></citation></ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>C.</given-names></name></person-group> (<year>2023</year>). <article-title>Research on emotion recognition-based smart assistant system: emotional intelligence and personalized services</article-title>. <source>J. Syst. Manag. Sci.</source> <volume>13</volume>, <fpage>227</fpage>&#x02013;<lpage>242</lpage>. <pub-id pub-id-type="doi">10.33168/JSMS.2023.0515</pub-id></citation>
</ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zogan</surname> <given-names>H.</given-names></name> <name><surname>Razzak</surname> <given-names>I.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Jameel</surname> <given-names>S.</given-names></name> <name><surname>Xu</surname> <given-names>G.</given-names></name></person-group> (<year>2022</year>). <article-title>Explainable depression detection with multi-aspect features using a hybrid deep learning model on social media</article-title>. <source>World Wide Web</source> <volume>25</volume>, <fpage>281</fpage>&#x02013;<lpage>304</lpage>. <pub-id pub-id-type="doi">10.1007/s11280-021-00992-2</pub-id><pub-id pub-id-type="pmid">35106059</pub-id></citation></ref>
</ref-list>
</back>
</article>