<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" dtd-version="1.3" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2025.1665798</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Research</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>A self-learning multimodal approach for fake news detection</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Chen</surname> <given-names>Hao</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Funding acquisition" vocab-term-identifier="https://credit.niso.org/contributor-roles/funding-acquisition/">Funding acquisition</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<uri xlink:href="https://loop.frontiersin.org/people/3129673"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Yu</surname> <given-names>Yue</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Guo</surname> <given-names>Hui</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Hu</surname> <given-names>Baochen</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Hu</surname> <given-names>Shu</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<uri xlink:href="https://loop.frontiersin.org/people/3077940"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Hu</surname> <given-names>Jinrong</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Visualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<uri xlink:href="https://loop.frontiersin.org/people/2659997"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Lyu</surname> <given-names>Siwei</given-names></name>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Wu</surname> <given-names>Xi</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Lin</surname> <given-names>Ching-Sheng</given-names></name>
<xref ref-type="aff" rid="aff7"><sup>7</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Wang</surname> <given-names>Xin</given-names></name>
<xref ref-type="aff" rid="aff8"><sup>8</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<uri xlink:href="https://loop.frontiersin.org/people/3078615"/>
</contrib>
</contrib-group>
<aff id="aff1"><label>1</label><institution>School of Computer Science, Chengdu University of Information Technology</institution>, <city>Chengdu</city>, <country country="cn">China</country></aff>
<aff id="aff2"><label>2</label><institution>CAACSRI</institution>, <city>Chengdu</city>, <country country="cn">China</country></aff>
<aff id="aff3"><label>3</label><institution>Department of Mathematics &#x00026; Statistics, University at Albany</institution>, <city>New York, NY</city>, <country country="us">United States</country></aff>
<aff id="aff4"><label>4</label><institution>Dropbox Inc.</institution>, <city>California, CA</city>, <country country="us">United States</country></aff>
<aff id="aff5"><label>5</label><institution>School of Applied and Creative Computing, Purdue University</institution>, <city>West Lafayette, IN</city>, <country country="us">United States</country></aff>
<aff id="aff6"><label>6</label><institution>Department of Computer Science and Engineering, University of Buffalo</institution>, <city>New York, NY</city>, <country country="us">United States</country></aff>
<aff id="aff7"><label>7</label><institution>Master Program of Digital Innovation, Tunghai University</institution>, <city>Taichung</city>, <country country="tw">Taiwan</country></aff>
<aff id="aff8"><label>8</label><institution>Department of Epidemiology and Biostatistics, College of Integrated Health Sciences, and AI Plus Institute, University at Albany</institution>, <city>New York, NY</city>, <country country="us">United States</country></aff>
<author-notes>
<corresp id="c001"><label>&#x0002A;</label>Correspondence: Jinrong Hu, <email xlink:href="mailto:hjr@cuit.edu.cn">hjr@cuit.edu.cn</email></corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2025-11-06">
<day>06</day>
<month>11</month>
<year>2025</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>8</volume>
<elocation-id>1665798</elocation-id>
<history>
<date date-type="received">
<day>14</day>
<month>07</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>20</day>
<month>10</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Chen, Yu, Guo, Hu, Hu, Hu, Lyu, Wu, Lin and Wang.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Chen, Yu, Guo, Hu, Hu, Hu, Lyu, Wu, Lin and Wang</copyright-holder>
<license>
<ali:license_ref start_date="2025-11-06">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<p>The rapid growth of social media has resulted in an explosion of online news content, leading to a significant increase in the spread of misleading or false information. While machine learning techniques have been widely applied to detect fake news, the scarcity of labeled datasets remains a critical challenge. Misinformation frequently appears as paired text and images, where a news article or headline is accompanied by a related visuals. In this paper, we introduce a self-learning multimodal model for fake news classification. The model leverages contrastive learning, a robust method for feature extraction that operates without requiring labeled data, and integrates the strengths of Large Language Models (LLMs) to jointly analyze both text and image features. LLMs are excel at this task due to their ability to process diverse linguistic data drawn from extensive training corpora. Our experimental results on a public dataset demonstrate that the proposed model outperforms several state-of-the-art classification approaches, achieving over 85% accuracy, precision, recall, and F1-score. These findings highlight the model&#x00027;s effectiveness in tackling the challenges of multimodal fake news detection.</p></abstract>
<kwd-group>
<kwd>fake news</kwd>
<kwd>contrastive learning</kwd>
<kwd>large language model</kwd>
<kwd>multimodal</kwd>
<kwd>machine learning</kwd>
</kwd-group>
<funding-group>
<funding-statement>The author(s) declare that financial support was received for the research and/or publication of this article. This research was funded by the Chengdu University of Information Technology Program (No. KYTZ2023053).</funding-statement>
</funding-group>
<counts>
<fig-count count="4"/>
<table-count count="4"/>
<equation-count count="7"/>
<ref-count count="37"/>
<page-count count="10"/>
<word-count count="7492"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>AI for Human Learning and Behavior Change</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<label>1</label>
<title>Introduction</title>
<p>The emergence of social media platforms has profoundly transformed news dissemination, offering immediate and widespread access to diverse information (<xref ref-type="bibr" rid="B34">Wang et al., 2023</xref>). However, this increased accessibility has inadvertently facilitated the rapid spread of misinformation, a problem exacerbated by technologies such as deepfakes (<xref ref-type="bibr" rid="B5">Chadha et al., 2021</xref>). A piece of misinformation is illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref>, where it is evident that the images are paired with misleading textual content, which can be easily fabricated or manipulated using AI-driven tools (<xref ref-type="bibr" rid="B12">Guo et al., 2022a</xref>). As a result, such misinformation can rapidly spread across the vast expanse of the digital landscape. In recent years, the widespread distribution of false narratives has become a critical social issue, causing negative impacts within digital environments and broader social contexts. This situation has raised substantial concerns across various demographic groups, leading to increased anxiety and a decrease in public trust in media sources (<xref ref-type="bibr" rid="B26">Pu et al., 2022</xref>). Therefore, there is an urgent need for the development and implementation of effective detection systems to combat the spread of fake news on social media platforms.</p>
<fig position="float" id="F1">
<label>Figure 1</label>
<caption><p>An example of fake news (mismatching image-text) from dataset (reproduced from <xref ref-type="bibr" rid="B23">Nakamura et al., 2020</xref>, European Language Resources Association (ELRA), licensed under CC-BY-NC).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1665798-g0001.tif">
<alt-text content-type="machine-generated">Cars drive along a snowy road toward a dramatic sunset with a vivid orange sky, creating an impression similar to a nuclear explosion. Silhouettes of city buildings are visible in the distance.</alt-text>
</graphic>
</fig>
<p>The laborious and time-consuming nature of manual fact-checking has spurred the development of automated approaches to address the widespread issue of fake news. Among these, machine learning techniques-particularly supervised classification models (<xref ref-type="bibr" rid="B2">Bagozzi et al., 2024</xref>; <xref ref-type="bibr" rid="B17">Kaliyar et al., 2020</xref>; <xref ref-type="bibr" rid="B15">Jiang et al., 2020</xref>) have attracted significant attention. However, the efficacy of these models largely depends on the availability of high-quality labeled datasets. Unfortunately, such datasets are often challenging to obtain and are typically insufficient to capture the full diversity inherent in fake news content due to their limited scope. In contrast, using weakly supervised or unsupervised methods mitigates the need for large volumes of labeled data and offers distinct advantages over traditional supervised approaches.</p>
<p>While existing approaches primarily focus on textual semantic and syntactic similarities and have achieved some success in fake news detection, they often fail to account for the intricate interactions between different data modalities, particularly the subtle and complex relationships between images and text. Currently, Large multimodal models (e.g., Gemini, GPT) demonstrate strong image&#x02013;text understanding through extensive pre-training, but their computational cost and general-purpose design limit applicability in domain-specific tasks such as fake news detection. In this paper, we propose a novel methodology for multi-modal fake news detection to address these shortcomings. Our approach uses contrastive learning to mitigate the challenge of limited labeled data. Additionally, we incorporate a large language model to integrate and analyze image and text features, enhancing the model&#x00027;s ability to assess the integrity of information. This design enables effective multimodal reasoning on smaller datasets while allowing targeted fine-tuning for domain-relevant cues. Hence, our approach provides a practical balance between performance and accessibility, complementing rather than competing with large generalist models. The primary contribution of this work lies in the architectural integration and synergistic design of several advanced yet complementary techniques to enhance multimodal fake news detection. Specifically:</p>
<list list-type="order">
<list-item><p>Incorporate contrastive learning into the visual feature extraction pipeline, enabling the model to better capture image semantics and improve generalization. This architectural choice increase detection performance, particularly under conditions of limited labeled data.</p></list-item>
<list-item><p>Build upon existing large language model (LLM) architectures, we design a framework that integrates textual and visual modalities through learnable queries and prompt-based alignment. This strategic combination allows for more coherent multimodal reasoning, leading to notably higher detection accuracy.</p></list-item>
<list-item><p>Design a dynamic optimization strategy for the loss function that adapts to the evolving state of the LLM during fine-tuning. This mechanism ensures stable convergence and maintains high detection performance throughout the training process.</p></list-item>
</list>
<p>It is important to note that most of the techniques we introduce are general and can be applied to various classification tasks. Specifically, our use of contrastive learning proves advantageous in scenarios with a scarcity of labeled training data. The paper is organized as follows: Section 2 reviews related work, and Section 3 describes our proposed method. Our experimental evaluation is presented in Section 4. The conclusion and future work are presented in Section 6.</p></sec>
<sec id="s2">
<label>2</label>
<title>Related work</title>
<p>In recent years, the widespread dissemination of fake news on social media platforms has resulted in significant detrimental effects, thereby motivating the scholarly community to engage in extensive investigations into fake news detection. Initially, <xref ref-type="bibr" rid="B3">Bondielli and Marcelloni (2019)</xref> conducted a rigorous classification of information, distinguishing between fake news and rumors based on whether credible sources had thoroughly verified the content. Subsequently, both <xref ref-type="bibr" rid="B22">Meel and Vishwakarma (2020)</xref> and <xref ref-type="bibr" rid="B11">Guo et al. (2019)</xref> provided comprehensive analyses of the various terminologies related to misinformation prevalent on social media, including disinformation, fake news, and misinformation, among others (<xref ref-type="bibr" rid="B13">Guo et al., 2022b</xref>). Instead of concentrating on the subtle and complex distinctions among these definitions, our research is primarily focused on the machine learning methodologies employed for detection. Furthermore, our study predominantly examines news articles that feature paired text and images. Within this framework, the present paper categorizes existing fake news detection techniques into two main types&#x02014;unimodal and multimodal&#x02014;based on the nature of the data utilized.</p>
<sec>
<label>2.1</label>
<title>Unimodal classification</title>
<p>A substantial body of research has leveraged supervised learning algorithms, such as Support Vector Machines (SVM) (<xref ref-type="bibr" rid="B2">Bagozzi et al., 2024</xref>), Na&#x000EF;ve Bayes (<xref ref-type="bibr" rid="B10">Granik and Mesyura, 2017</xref>), and Logistic Regression (<xref ref-type="bibr" rid="B32">Sudhakar and Kaliyamurthie, 2023</xref>), for the detection of fake news. These models are trained on annotated datasets that classify news articles as either authentic or deceptive. The performance of these algorithms, however, is highly contingent on the quality and diversity of the training data provided. Moving beyond conventional machine learning methods, neural networks have gained prominence in this domain (<xref ref-type="bibr" rid="B14">Guo et al., 2022c</xref>). For instance, convolutional neural networks (CNN) were employed by <xref ref-type="bibr" rid="B17">Kaliyar et al. (2020)</xref>, recurrent neural networks (RNN) were used by <xref ref-type="bibr" rid="B15">Jiang et al. (2020)</xref>, and <xref ref-type="bibr" rid="B24">Nasir et al. (2021)</xref> implemented a hybrid of CNN and RNN. Recently, researchers have explored pre-trained language models. BERT (<xref ref-type="bibr" rid="B8">Devlin et al., 2019</xref>) and RoBERTa (<xref ref-type="bibr" rid="B37">Zhuang et al., 2021</xref>) are employed to analyze news content, achieving notable advancements. Nevertheless, due to their relatively modest model sizes, the capacity of these pre-trained models to extract complex knowledge and perform advanced reasoning is constrained, limiting their effectiveness in handling fake news that requires deeper content analysis and inference. In contrast, large language models (LLMs), such as GPT (<xref ref-type="bibr" rid="B4">Brown et al., 2020</xref>), exhibit superior performance in natural language processing (NLP) tasks by employing deeper neural architectures and significantly larger parameter counts. These models rely on extensive textual datasets from diverse fields and topics, constructing a comprehensive knowledge base and contextual understanding that bolsters their reasoning capabilities. Consequently, LLMs require minimal additional data for fine-tuning to effectively differentiate authentic news from misinformation (<xref ref-type="bibr" rid="B20">Li et al., 2024</xref>; <xref ref-type="bibr" rid="B31">Su et al., 2023</xref>; <xref ref-type="bibr" rid="B33">Teo et al., 2024</xref>). However, most existing studies have predominantly focused on either text or image data in isolation, rather than integrating both modalities for a more comprehensive approach.</p>
</sec>
<sec>
<label>2.2</label>
<title>Multimodal classification</title>
<p>In the wake of the incessant development of social media, the news that is now being widely spread predominantly incorporates information such as text and images. As a result, scholars have stepped up their endeavors in the detection of multimodal fake news. <xref ref-type="bibr" rid="B30">Singhal et al. (2019)</xref> introduced a multimodal model which harnesses text and visual features. Likewise, <xref ref-type="bibr" rid="B9">Giachanou et al. (2020)</xref> combined the image features extracted by the VGG (<xref ref-type="bibr" rid="B29">Simonyan and Zisserman, 2014</xref>) model and the text features extracted by the BERT (<xref ref-type="bibr" rid="B8">Devlin et al., 2019</xref>) model to detect image-text misinformation. <xref ref-type="bibr" rid="B1">Aneja et al. (2022)</xref> centered their attention on &#x0201C;Cheapfakes&#x0201D; produced by employing free artificial intelligence methods (e.g., filtering). They made use of multimodal embedding to predict whether image-caption pairs are mismatched. Fact verification, on the other hand, necessitates an additional information base. In light of this circumstance, some scholars have focused on the use of LLMs with pre-trained on the prior knowledge. <xref ref-type="bibr" rid="B36">Zhu et al. (2023)</xref> introduced the multimodal large language model MiniGPT-4, which achieves alignment between image and linguistic features by employing the Q-Former module. <xref ref-type="bibr" rid="B21">Liu et al. (2024)</xref> introduced a fake news detection model, FakeNewsGPT-4, leveraging the MiniGPT-4 framework. This model advances the prompting capabilities of large language models by integrating both prior and dynamically generated knowledge, thereby achieving superior performance across multiple domains.</p></sec>
</sec>
<sec sec-type="methods" id="s3">
<label>3</label>
<title>Methodology</title>
<p>In this paper, we propose an innovative model for the task of multimodal fake news detection, as illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref>. The overall structure comprises three core components: the contrastive module, the multimodal fusion module, and the classification module. To overcome the lack of training data, a contrastive learning mechanism is incorporated. In this mechanism, the image feature is acquired through augmentation where the model is trained by maximizing the similarity between positive pairs (e.g., different augmentations of the same data instance) and minimizing the similarity between negative pairs (from different data instances). Once the image features has been learned, we then use the pre-trained large-scale model to align the text content with the image features, which we called the infuse module. Rather than directly using the image features extracted from contrastive learning, we introduced a multi-task learning approach namely Q-Former to dynamically adjust the feature weights of image, thus to achieve the co-optimization of the image feature encoder and the text encoder. Subsequently, the pre-trained large-scale model, which is structured upon MiniGPT-4, is employed to effect a profound combination of image and text features. The multimodal features along with the appropriate prompts are fed as input of large language model. The output is followed by the linear layer and finally is trained to classify an accurate inference regarding the authenticity of news. We will explain each module in detail as follows:</p>
<fig position="float" id="F2">
<label>Figure 2</label>
<caption><p>The overall structure of multimodal fake news detection (images reproduced from <xref ref-type="bibr" rid="B23">Nakamura et al., 2020</xref>, the Fakeddit dataset, <ext-link ext-link-type="uri" xlink:href="https://github.com/entitize/Fakeddit">https://github.com/entitize/Fakeddit</ext-link>). The model is composed of three components, contrastive learning module is for learning the image feature using a small sample of training data, infusing module aims to align text and image feature and then apply the large language model for the multimodal combination, the classification module is for the prediction of fake news.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1665798-g0002.tif">
<alt-text content-type="machine-generated">Diagram illustrating a machine learning process divided into three modules. The Contrastive Learning Module shows image mini-batches undergoing augmentation, feature extraction, and forming a key-value matrix for InfoNCE loss. The Multimodal Infuse Module combines text tokenization with a Q-Former and LLM. The Classification Module includes a linear layer and cross-entropy loss for binary classification.</alt-text>
</graphic>
</fig>
<sec>
<label>3.1</label>
<title>The contrastive learning module</title>
<p>Within the contrastive learning module, each of the image undergoes augmentation procedures. Subsequently, the augmented images are inputted into an image encoder, which is then followed by a fully connected layer. This sequential process is designed to learn the image feature and generate a dense vector. For the purpose of training the contrastive learning model, all the augmented images originating from the same sample are regarded as positive instances. Meanwhile, images that are randomly selected from other samples within the dataset are considered as negative instances. To be specific, the image is augmented (e.g., rotation, flip, scaling etc.) and then fed to the encoder for feature extraction. We adopt the pre-trained ViT model (<xref ref-type="bibr" rid="B10">Granik and Mesyura, 2017</xref>) as the backbone network.</p>
<p>Rather than directly training ViT for image feature extraction, we harness the power of momentum mechanism to smooth and stabilize the image encoding. As shown in <xref ref-type="fig" rid="F3">Figure 3</xref>, the input image <italic>Z</italic>, along with the augmentation <inline-formula><mml:math id="M1"><mml:mover accent="true"><mml:mrow><mml:mi>Z</mml:mi></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:math></inline-formula>, are fed to the the image encoder and momentum encoder. Both of them are identical initially while performing training procedure individually. The difference is that the parameters of the momentum encoder are not updated synchronously with those of the encoder. Updating the momentum encoder parameters using a smaller value ensures the stability of the training process, as shown in <xref ref-type="disp-formula" rid="EQ1">Equation 1</xref>:</p>
<disp-formula id="EQ1"><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>m</mml:mi><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(1)</label></disp-formula>
<p>where <italic>y</italic><sub><italic>t</italic>&#x02212;1</sub> is the parameter at the previous moment, <italic>x</italic><sub><italic>t</italic></sub> is the parameter at the current moment, and <italic>m</italic> is the updated threshold of the parameter. The image features from two encoders are conducted scalar product respectively to get two matrices. Then, the two matrices are added to get final matrix for contrastive learning. We use mini-batch for training. The elements of the diagonal of the matrix indicate the similarity between the positive samples within a mini-batch, and the other elements indicate the similarity between the positive samples and the negative samples. The overall loss function of contrastive learning is using InfoNCE as shown in <xref ref-type="disp-formula" rid="EQ2">Equation 2</xref>:</p>
<disp-formula id="EQ2"><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mo class="qopname">log</mml:mo><mml:mfrac><mml:mrow><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>q</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:msup><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002B;</mml:mo></mml:mrow></mml:msup><mml:mo>/</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>q</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:msup><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002B;</mml:mo></mml:mrow></mml:msup><mml:mo>/</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:mstyle><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>q</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:msup><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo></mml:mrow></mml:msup><mml:mo>/</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(2)</label></disp-formula>
<p>where &#x003C4; is the temperature coefficient that controls the softness or sharpness of the probability distribution over the positive and negative samples when calculating the InfoNCE loss. <italic>q</italic> is an anchor sample, <italic>k</italic><sup>&#x0002B;</sup> is the positive sample, and <italic>k</italic><sup>&#x02212;</sup> is the negative sample. Mathematically, the InfoNCE loss is given by a formula that involves taking the logarithm of the ratio of the exponential of the score of the positive sample to the sum of the exponentials of the scores of all samples (both positive and negative).</p>
<fig position="float" id="F3">
<label>Figure 3</label>
<caption><p>Momentum configuration for contrastive learning (image reproduced from <xref ref-type="bibr" rid="B23">Nakamura et al., 2020</xref>, the Fakeddit dataset, <ext-link ext-link-type="uri" xlink:href="https://github.com/entitize/Fakeddit">https://github.com/entitize/Fakeddit</ext-link>).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1665798-g0003.tif">
<alt-text content-type="machine-generated">Diagram illustrating a machine learning process with two sets of images input into an Encoder and a Momentum Encoder. Outputs, represented as vectors, are compared through matrix operations labeled &#x0201C;positive&#x0201D;. Connections indicate the flow of data towards a final matrix at the top.</alt-text>
</graphic>
</fig>
</sec>
<sec>
<label>3.2</label>
<title>Multimodal fusion module</title>
<p>Concurrently, with regard to the text content, the Byte-Pair Encoding algorithm is initially employed as a tokenizer for the purpose of transforming sentences into tokens. Subsequently, the multimodal model predicated on MiniGPT-4 amalgamates the text along with the image features that have been encoded during the pre-training phase. Given that the large language model (LLM) utilized in our paper has already undergone pre-training with an extensive volume of data in advance, the downstream task merely necessitates fine-tuning with a relatively small quantity of data to fulfill the specified task requirements.</p>
<p>Instead of using the image features from contrastive learning module, we would like to leverage Q-Former (Query Transformer) (<xref ref-type="bibr" rid="B19">Li et al., 2023</xref>), a core component of MiniGPT-4 (<xref ref-type="bibr" rid="B23">Nakamura et al., 2020</xref>), to bridge the relationship between image and text. It employs a small set of learnable query tokens that interact with visual encoder outputs through cross-attention mechanisms, enabling the extraction of semantically relevant features. The resulting query embeddings form a compact and interpretable representation that is compatible with large language models, thereby facilitating efficient and effective multimodal fusion. In detail, the module extracts features from a frozen image encoder and aligns them with a large language model. The key component is a learnable query which is a set of vectors that are designed to interact with the input data in a way that helps extract relevant information and establish meaningful connections. The structure is illustrated in <xref ref-type="fig" rid="F4">Figure 4</xref>. Q-Former&#x00027;s alignment is divided into two phases, the first phase mainly involves the learning of image features, through which Q-Former learns the most relevant image feature representation to the current text by the learnable queries. The second stage is generative learning, which combines the output of Q-Former with a pre-trained frozen large language model to achieve visual-to-language generative learning. The LLM [Vicuna (<xref ref-type="bibr" rid="B7">Chiang et al., 2023</xref>)] is used to understand and describe the visual expression features of the Q-Former output, thus building a relationship between visual information and linguistic description.</p>
<fig position="float" id="F4">
<label>Figure 4</label>
<caption><p>Q-Former structure adopted from <xref ref-type="bibr" rid="B19">Li et al. (2023)</xref>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1665798-g0004.tif">
<alt-text content-type="machine-generated">Diagram showing the architecture of a system involving an Image Encoder and a Q-Former. The Q-Former processes image representation through self-attention, cross-attention, and feed-forward layers with learnable queries. The output connects to a Large Language Model (LLM), which also receives text and prompts.</alt-text>
</graphic>
</fig>
<p>In the end of multimodal fusion module, text feature <italic>e</italic><sub><italic>text</italic></sub> is obtained through the text embedding layer, image feature <italic>e</italic><sub><italic>img</italic></sub> is obtained in a pre-trained image encoder, and function <italic>f</italic> that can be interpreted by the large language model are obtained by using Q-Former with approapriate prompt, and these features are connected together to obtain the final mixed feature representation <italic>E</italic>. The equations are shown in <xref ref-type="disp-formula" rid="EQ3">Equations 3</xref>, <xref ref-type="disp-formula" rid="EQ4">4</xref>.</p>
<disp-formula id="EQ3"><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">prompt</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">text</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>Q</mml:mi><mml:mo>-</mml:mo><mml:mtext class="textrm" mathvariant="normal">Former</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">img</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(3)</label></disp-formula>
<disp-formula id="EQ4"><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mi>L</mml:mi><mml:mi>L</mml:mi><mml:mi>M</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">prompt</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(4)</label></disp-formula>
</sec>
<sec>
<label>3.3</label>
<title>Classification module</title>
<p>After performing multimodal fusion module to obtain the hidden layer features <italic>E</italic>, these features are then input into the fake news classifier. Subsequently, the final authenticity of the news is output. The fake news classifier is composed of two linear layers along with a GELU activation function. The specific details of the classification process are presented in <xref ref-type="disp-formula" rid="EQ5">Equation 5</xref> below:</p>
<disp-formula id="EQ5"><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtable><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>M</mml:mi><mml:mi>L</mml:mi><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>E</mml:mi><mml:mo>&#x02223;</mml:mo><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(5)</label></disp-formula>
<p>where &#x003B8; is all the model parameters and <italic>l</italic> is the specific news category which is true or false.</p>
</sec>
<sec>
<label>3.4</label>
<title>Prompt</title>
<p>Within the LLM domain, prompts hold a vital position as they significantly influence the model&#x00027;s actions and resultant outputs. Essentially, prompts are text-based instructions or input indications furnished to the LLM, with the aim of communicating the task or context that the model is required to manage. In our task, we crafted a variety of prompts allow the model be able to differentiate the authentic of news from different perspectives. The methodology put forward within this chapter employs four distinct Prompts for the training of the model. These Prompts are designed in line with the session format of the Vicuna language model and are randomly chosen to be inputted into the model. Sepecifically, the prompt is beginning with &#x0003C; <italic>Img&#x0003E; &#x0003C; ImageHere&#x0003E; &#x0003C; /Img&#x0003E; &#x0003C; Text&#x0003E; &#x0003C; TextHere&#x0003E; &#x0003C; /Text&#x0003E;</italic>, where &#x0003C; <italic>ImageHere&#x0003E;</italic> represents the visual features generated by the image encoder, while &#x0003C; <italic>TextHere&#x0003E;</italic> represents the corresponding news text content. It followed by the different questions as follows:</p>
<list list-type="bullet">
<list-item><p>Determine if the text and images of this news story are lying?</p></list-item>
<list-item><p>Determine if this text and images are describing a rumor?</p></list-item>
<list-item><p>Determine if this news story is false?</p></list-item>
<list-item><p>Determine if this text and image information is untrue?</p></list-item>
</list>
</sec>
<sec>
<label>3.5</label>
<title>Loss function</title>
<p>In the model previously described, both the contrastive learning and classification components necessitate training procedures. As a result, the overall loss function is constituted by two sub-functions, namely the contrastive learning loss <italic>L</italic><sub>1</sub> and the classification loss <italic>L</italic><sub>2</sub>. Given the varying optimization priorities between these two individual tasks, employing fixed weights does not yield optimal results, and manually adjusting these weights is time-consuming. To address this issue, we implemented the Automatic Weighted Loss (AWL) method <xref ref-type="bibr" rid="B18">Kendall et al. (2018)</xref> to augment multi-task learning and concurrently optimize multiple loss functions. The details of this implementation are illustrated in <xref ref-type="disp-formula" rid="EQ6">Equation 6</xref>.</p>
<disp-formula id="EQ6"><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>L</mml:mi><mml:mo>&#x02248;</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:mfrac><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:mfrac><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(6)</label></disp-formula>
<p>Where &#x003C3;<sub>1</sub> and &#x003C3;<sub>2</sub> are learnable parameters representing the uncertainty of the corresponding task, the higher the uncertainty of the task, the smaller the weight of its loss function. Through this dynamic adjustment, the model can learn the weights suitable for different tasks and avoid the negative impact of fine-tune of pre-trained model. Additionally, during the preliminary experiment, we found that the loss function was getting smaller, but the model was not improved. To avoid this issue, we add 1 to each of learnable parameters. To be clarified, we train contrastive learning module based on <italic>L</italic><sub>1</sub> for all images in the dataset. Then we fix the contrastive learning module and train <italic>L</italic><sub>2</sub>. The weights of the whole model are updated iteratively as described in <xref ref-type="bibr" rid="B6">Chen et al. (2023)</xref>.</p></sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Experimental settings</title>
<sec>
<label>4.1</label>
<title>Datasets</title>
<p>The data (<xref ref-type="bibr" rid="B23">Nakamura et al., 2020</xref>) obtained from social media platforms generally comprises a variety of elements. Among them, invalid information such as URLs and stop words are frequently encountered. This study aims to remove such information. However, certain proper names, including those of individuals, locations, and countries, which are considered as key information, are deliberately retained on account of the implementation of LLMs. For the image data, a series of augmentation operations are meticulously devised. These encompass operations such as horizontal flipping, hue transformation, and grayscale conversion. During the training phase of the model, one of these augmentation operations is randomly selected and applied. In addition, the data associated with entries lacking attached images and those accompanied by invalid images are removed from the dataset. Following this pre-processing, a complete and refined dataset is constructed. This dataset comprises nearly 563,600 training samples, 59,000 validation samples, and 59,500 testing samples. The statistical details of the dataset are presented in <xref ref-type="table" rid="T1">Table 1</xref> below.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Statistical information of the dataset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Data category</bold></th>
<th valign="top" align="center"><bold>True</bold></th>
<th valign="top" align="center"><bold>False</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Training set</td>
<td valign="top" align="center">222.1k</td>
<td valign="top" align="center">341.5k</td>
</tr> <tr>
<td valign="top" align="left">Validation set</td>
<td valign="top" align="center">23k</td>
<td valign="top" align="center">36k</td>
</tr> <tr>
<td valign="top" align="left">Testing set</td>
<td valign="top" align="center">23.5k</td>
<td valign="top" align="center">36k</td>
</tr></tbody>
</table>
</table-wrap>
</sec>
<sec>
<label>4.2</label>
<title>Implementation details</title>
<p>The image encoder of our network employed off-the-shelf ViT model where its block size is set to 14. All input images were scaled to a fixed resolution of 224 &#x000D7; 224. Followed by the encoding process, the feature vector dimension was set up to the batch size &#x000D7; 12 &#x000D7; 1,048. In terms of large language model we used, it is worth noting that the Vicuna model applied in this paper is version 7B. During the training process, a total of 100 training epochs were conducted, which we proved it&#x00027;s the best practices based on a variety of preliminary experiments, we extended the number of samples in each batch to 96 by the gradient accumulation method. The assessment methods in this paper are chosen as the accuracy, precision, recall, and F1-score, which are widely used in the field of fake news detection, as the metrics for evaluating the performance of the model.</p>
</sec>
<sec>
<label>4.3</label>
<title>Baselines</title>
<p>To assess the efficacy of our proposed model, this study selects a series of state-of-the-art classification methods as the baselines for comparing. All the chosen models are widely employed in image-text paring tasks and achieved strong ability to detect fake news. All models were trained using identical publicly available datasets and tested on the same data. Various evaluation metrics were carried out. Furthermore, we investigated model performance under conditions of limited training data, specifically utilizing only 10% of the Fakeddit training set. The baseline methods are: <bold>EANN</bold> (<xref ref-type="bibr" rid="B35">Wang et al., 2018</xref>) employs the pre-trained VGG for extracting image features, followed by Text-CNN. EANN incorporates auxiliary tasks such as an event discriminator, which outputs event categories of news to aid in decision-making. <bold>CAFE</bold> (<xref ref-type="bibr" rid="B16">Jin et al., 2021</xref>) leverages pre-trained BERT and ResNet-34 models to extract features from text and image respectively. It used a unified embedding space in order to integrate diverse modalities. <bold>SpotFake</bold> (<xref ref-type="bibr" rid="B30">Singhal et al., 2019</xref>) utilizes a pre-trained VGG network to extract features from images and BERT to capture textual features. SpotFake&#x00027;s notable advantage lies in its simplicity, avoiding complex auxiliary training tasks often seen in other models. <bold>SpotFake&#x0002B;</bold> (<xref ref-type="bibr" rid="B30">Singhal et al., 2019</xref>) modified SpotFake framework by the use of additional layers with attention mechanisms, advanced regularization techniques, and more sophisticated training process. <bold>MVAE</bold> (<xref ref-type="bibr" rid="B27">Qi et al., 2019</xref>) The Multimodal Variational Autoencoder extends variational autoencoders for multimodal data (text, images, audio). It learns shared latent representations to address tasks like classification. <bold>HMCAN</bold> (<xref ref-type="bibr" rid="B28">Qian et al., 2021</xref>) is short for Hierarchical Memory Compressed Attention Network which introduces a hierarchical structure with memory-compressed attention to capture both local and global contexts, optimizing long-sequence processing in NLP tasks. <bold>VERITE</bold> (<xref ref-type="bibr" rid="B25">Papadopoulos et al., 2023</xref>): &#x0201C;VERification of Image-TExt pairs&#x0201D; uses CLIP (Contrastive Language&#x02013;Image Pre-training) as the feature extractor for images and texts, followed by the encoding layer of the Transformer. It has demonstrated a strong fake news detection performance.</p></sec>
</sec>
<sec sec-type="results" id="s5">
<label>5</label>
<title>Results</title>
<sec>
<label>5.1</label>
<title>General performance</title>
<p>The <xref ref-type="table" rid="T2">Table 2</xref> presents a comparative analysis of various models based on four evaluation metrics: accuracy, precision, recall, and F1-score. The models compared include EANN, CAFE, SpotFake, SpotFake&#x0002B;, MVAE, HMCAN, VERITE, and the proposed model referred to as &#x0201C;Ours.&#x0201D; EANN achieved the lowest accuracy of 72.27% and an F1-score of 70.12%. In contrast, CAFE performed significantly better, with an accuracy of 84.14% and an F1-score of 85.32%, indicating robust performance in both precision and recall. Comparing SpotFake and its enhanced version, SpotFake&#x0002B; showed notable improvement with SpotFake&#x0002B; reaching an accuracy of 83.08% and an F1-score of 85.62%, outperforming SpotFake&#x00027;s 77.29% accuracy and 71.20% F1-score. HMCAN demonstrated superior precision, recall, and F1-score values, all around 84%, compared to MVAE&#x00027;s lower performance in these metrics. VERITE model also showed high performance with an accuracy of 84.72% and an F1-score of 84.85%, closely aligning with CAFE and SpotFake&#x0002B; in terms of overall effectiveness.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>The results of three models over the Accuracy, Precision, Recall and F1-score.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="center"><bold><italic>Accuracy</italic></bold></th>
<th valign="top" align="center"><bold><italic>Precision</italic></bold></th>
<th valign="top" align="center"><bold><italic>Recall</italic></bold></th>
<th valign="top" align="center"><bold><italic>F1-Score</italic></bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><italic><bold>EANN</bold></italic> (<xref ref-type="bibr" rid="B35">Wang et al., 2018</xref>)</td>
<td valign="top" align="center">72.27</td>
<td valign="top" align="center">78.43</td>
<td valign="top" align="center">63.4</td>
<td valign="top" align="center">70.12</td>
</tr> <tr>
<td valign="top" align="left"><italic><bold>CAFE</bold></italic> (<xref ref-type="bibr" rid="B16">Jin et al., 2021</xref>)</td>
<td valign="top" align="center">84.14</td>
<td valign="top" align="center">85.39</td>
<td valign="top" align="center">85.27</td>
<td valign="top" align="center">85.32</td>
</tr> <tr>
<td valign="top" align="left"><italic><bold>SpotFake</bold></italic> (<xref ref-type="bibr" rid="B30">Singhal et al., 2019</xref>)</td>
<td valign="top" align="center">77.29</td>
<td valign="top" align="center">71.63</td>
<td valign="top" align="center">70.77</td>
<td valign="top" align="center">71.20</td>
</tr> <tr>
<td valign="top" align="left"><italic><bold>SpotFake&#x0002B;</bold></italic> (<xref ref-type="bibr" rid="B30">Singhal et al., 2019</xref>)</td>
<td valign="top" align="center">83.08</td>
<td valign="top" align="center">86.38</td>
<td valign="top" align="center">84.87</td>
<td valign="top" align="center">85.62</td>
</tr> <tr>
<td valign="top" align="left"><italic><bold>MVAE</bold></italic> (<xref ref-type="bibr" rid="B27">Qi et al., 2019</xref>)</td>
<td valign="top" align="center">70.24</td>
<td valign="top" align="center">76.53</td>
<td valign="top" align="center">74.75</td>
<td valign="top" align="center">75.63</td>
</tr> <tr>
<td valign="top" align="left"><italic><bold>HMCAN</bold></italic> (<xref ref-type="bibr" rid="B28">Qian et al., 2021</xref>)</td>
<td valign="top" align="center">82.89</td>
<td valign="top" align="center">84.03</td>
<td valign="top" align="center">84.04</td>
<td valign="top" align="center">84.03</td>
</tr> <tr>
<td valign="top" align="left"><italic><bold>VERITE</bold></italic> (<xref ref-type="bibr" rid="B25">Papadopoulos et al., 2023</xref>)</td>
<td valign="top" align="center">84.72</td>
<td valign="top" align="center">85.34</td>
<td valign="top" align="center">84.37</td>
<td valign="top" align="center">84.85</td>
</tr> <tr>
<td valign="top" align="left"><italic><bold>Ours</bold></italic></td>
<td valign="top" align="center"><bold>88.88</bold></td>
<td valign="top" align="center"><bold>86.40</bold></td>
<td valign="top" align="center"><bold>85.40</bold></td>
<td valign="top" align="center"><bold>85.90</bold></td>
</tr></tbody>
</table>
</table-wrap>
<p>The proposed model (&#x0201C;Ours&#x0201D;) outperformed all baseline models with the highest accuracy of 88.88%, precision of 86.40%, recall of 85.40%, and F1-score of 85.90%. This indicates the superior capability of the our method in handling the experimental tasks, achieving a balanced and high performance across all evaluation metrics. The observed improvements in accuracy and recall of our model indicate a significant reduction in the likelihood of misclassifying real news articles as fake, as well as a substantial decrease in the incidence of false positives.</p>
<p>To further analyzing results, EANN and MVAE exhibit only slight performance differences, reflecting the limitations of their basic multimodal fusion approaches. EANN utilizes an event discriminator for classification but does not address the semantic relationships between text and image features. MVAE, while using decoding structures to improve performance, still struggles with misalignment between modalities. SpotFake outperforms both by using pre-trained BERT for text encoding, although the direct concatenation of features limits its potential. SpotFake&#x0002B; achieves even higher accuracy with XLNet, though it faces the same fusion constraints. HMCAN, CAFE, and VERITE outperform others, but fall short compared to the proposed method. HMCAN, about 6% less accurate than ours, applies multimodal contextual attention to fuse features. CAFE improves accuracy by aligning features in a unified space, reducing semantic gaps. VERITE stands out by combining CLIP for feature extraction with attention-based fusion, making it the top performer among the comparative models. To assess the effectiveness of the proposed method, statistical significance tests were performed across all evaluation metrics. The Friedman test indicated a significant overall difference among the eight compared models. Although subsequent Bonferroni-adjusted <italic>post-hoc</italic> comparisons did not reveal individually significant differences, the consistent gains observed in Accuracy, Precision, Recall, and F1-Score suggest a robust performance advantage. Notably, the proposed model achieved the highest Accuracy with a statistically significant margin (<italic>p</italic> &#x0003C; 0.05, one-tailed <italic>t</italic>-test), demonstrating superior effectiveness and stability relative to existing approaches. Based on the above experimental results, this paper believes that the main factors that affect classification performance are as follows:</p>
<list list-type="bullet">
<list-item><p>The approach of feature fusion exerts a substantial influence on the performance of the model. While the direct concatenation of features appears to be a straightforward method, it fails to facilitate the interaction among data from diverse modalities. In contrast, models engineered with feature fusion techniques tend to attain favorable accuracy levels.</p></list-item>
<list-item><p>Model scale also serves as a crucial metric. Apart from the model introduced in this study, the parameter scales of other comparative models are relatively limited. This smaller parameter size constrains the model&#x00027;s inferential abilities, impeding its capacity to make precise judgments regarding text and image features.</p></list-item>
</list>
</sec>
<sec>
<label>5.2</label>
<title>Ablation study</title>
<p>In this section, a series of ablation experiments were carried out. These experiments strictly adhered to the principle of controlling variables, ensuring that only one module was altered each time, thereby enabling the individual assessment of the impact of each module on the overall performance. Specifically, two types of experiments were designed, namely the comparison of different modal data and comparison of individual module. The specific settings of the experiments are as follows:</p>
<sec>
<label>5.2.1</label>
<title>Comparison of different modal data</title>
<p>To verify the impact of different modal data on fake news detection, three experiments were designed respectively, namely single-modal image, single-modal text, and the multimodal combination of image and text. According to the experimental results presented in <xref ref-type="table" rid="T3">Table 3</xref>, in the experiment where only single-modal image data was utilized, the lowest accuracy, precision, and F1-score were 79.68%, 68.76%, and 72.75% respectively. In contrast, the model using single-modal text data achieved relatively higher accuracy, precision, and F1-score, which were 83.18%, 79.71%, and 78.64% respectively. This indicates that text information is of greater significance in determining the authenticity of news compared to image information. The data of the multimodal combination of image and text yielded the highest accuracy of 88.88%, which is approximately 7% and 5% higher than that of single-modal image and text respectively. Moreover, the accuracy, recall rate, and F1-score were all at relatively high levels, thereby demonstrating the effectiveness of multimodal fusion in fake news detection.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Ablation comparison of different modals.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Modal type</bold></th>
<th valign="top" align="center"><bold>Accuracy</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>F1-score</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Image</td>
<td valign="top" align="center">79.68</td>
<td valign="top" align="center">68.76</td>
<td valign="top" align="center">89.38</td>
<td valign="top" align="center">72.75</td>
</tr> <tr>
<td valign="top" align="left">Text</td>
<td valign="top" align="center">83.18</td>
<td valign="top" align="center">79.71</td>
<td valign="top" align="center">77.25</td>
<td valign="top" align="center">78.46</td>
</tr> <tr>
<td valign="top" align="left">Image and text</td>
<td valign="top" align="center">88.88</td>
<td valign="top" align="center">86.40</td>
<td valign="top" align="center">85.40</td>
<td valign="top" align="center">85.90</td>
</tr></tbody>
</table>
</table-wrap>
</sec>
<sec>
<label>5.2.2</label>
<title>Comparison of individual modules</title>
<p>In order to precisely gauge the contributions of each module within the proposed methodology to the performance of the model, three sets of comparative experiments were meticulously designed for this ablation study. The detailed experimental configurations are presented as follows:</p>
<list list-type="bullet">
<list-item><p>Experiment A: Employed the large language model along with the fully connected layer.</p></list-item>
<list-item><p>Experiment B: Incorporated the contrastive learning module on top of the setup in Experiment A.</p></list-item>
<list-item><p>Experiment C: Added the multimodal infuse module based on the configuration of Experiment B.</p></list-item>
</list>
<p>As shown in <xref ref-type="table" rid="T4">Table 4</xref>, Experiment A displayed the weakest performance in the context of fake news detection. Its accuracy rate was merely 87.16%, which was lower than that attained by Experiment B and Experiment C. The models that utilized contrastive learning and multimodal learning approaches demonstrated an advantage over the LLM model across various metrics. The improvement could be ascribed to the data augmentation procedure, which broadened the data samples and thus improved the generalization and robustness of the model. When analyzing the precision and recall rates, it was found that the multi-task learning method maintained a relatively stable performance, while Experiments A and B showed relatively higher recall rates.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Ablation comparison of different modules.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="center"><bold>Accuracy</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>F1-score</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Experiment A</td>
<td valign="top" align="center">87.16</td>
<td valign="top" align="center">78.60</td>
<td valign="top" align="center">92.91</td>
<td valign="top" align="center">85.16</td>
</tr> <tr>
<td valign="top" align="left">Experiment B</td>
<td valign="top" align="center">88.21</td>
<td valign="top" align="center">81.19</td>
<td valign="top" align="center">91.45</td>
<td valign="top" align="center">86.02</td>
</tr> <tr>
<td valign="top" align="left">Experiment C</td>
<td valign="top" align="center">88.88</td>
<td valign="top" align="center">86.40</td>
<td valign="top" align="center">85.40</td>
<td valign="top" align="center">85.90</td>
</tr></tbody>
</table>
</table-wrap>
</sec></sec></sec>
<sec sec-type="conclusions" id="s6">
<label>6</label>
<title>Conclusion</title>
<p>This study aimed to establish an innovative multimodal classification framework for verifying the authenticity of news shared on social media platforms. In light of limited labeled data, especially for image-based content, we employed contrastive learning to improve feature representation. Additionally, we demonstrated the effectiveness of Large Language Models (LLMs) in facilitating the seamless integration of text and image features. Instead of a simplistic multimodal fusion, we introduced a learnable alignment module that significantly improved the model&#x00027;s accuracy by aligning text-image features. Key contributions of this work include: (1) We developed an enhanced fake news detection model grounded in contrastive learning, a self-supervised approach utilizing data augmentation during training. Experimental results strongly indicated the model&#x00027;s superiority in scenarios with limited labeled data. (2) Our focus on news items as paired image-text combinations revealed that dynamically infusing features from different data formats substantially improved fake news detection, achieving an accuracy rate close to 89%. This multimodal approach considerably outperformed single-modal text- or image-only analyses. (3) By integrating contrastive learning with LLMs through a carefully designed feature infusion mechanism for multimodal classification, we conducted extensive comparative experiments. The findings highlighted the robust detection capabilities of our proposed model relative to numerous existing models, and demonstrated that larger LLMs further enhance detection accuracy.</p>
<p>A primary limitation of this study is the scope of data analysis, as our current focus excludes potentially valuable supplementary information, such as social networks, geographic data, and event context. While our model achieves state-of-the-art results purely from the content, its full potential for misinformation detection on social media platforms remains untapped without these relational features. Furthermore, a significant limitation lies in the omission of computational efficiency metrics. Although our model demonstrates superior performance in Accuracy and F1-score, we did not report its run-time, inference speed, or the total computational resources (e.g., GPU hours) required for training. This prevents a complete evaluation of the crucial trade-off between performance and deployment cost. Future research will focus on two main directions: first, developing robust integration strategies to effectively combine our model with supplementary social and event-based information to further boost predictive accuracy; and second, conducting comprehensive optimization and benchmarking to reduce run-time, improve inference speed, and ensure the computational viability of our approach for real-world application.</p></sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://github.com/entitize/Fakeddit">https://github.com/entitize/Fakeddit</ext-link>.</p>
</sec>
<sec sec-type="ethics-statement" id="s8">
<title>Ethics statement</title>
<p>This study utilizes the publicly available Fakeddit dataset, which comprises Reddit posts collected in accordance with Reddit&#x00027;s content and API usage policies. All personal identifiers and user metadata were removed to ensure anonymity and prevent re-identification. Although no direct human participation was involved, data use adhered to privacy protection and data minimization principles consistent with responsible research standards. As Fakeddit reflects the biases of its source community, care was taken to mitigate potential effects on model fairness and interpretation. The research aims to advance academic understanding of misinformation detection, not to develop operational moderation systems. Ethical use was guided by transparency, accountability, and awareness of potential societal implications such as bias reinforcement or trust erosion.</p>
</sec>
<sec sec-type="author-contributions" id="s9">
<title>Author contributions</title>
<p>HC: Conceptualization, Writing &#x02013; original draft, Formal analysis, Funding acquisition, Methodology, Writing &#x02013; review &#x00026; editing. YY: Supervision, Writing &#x02013; original draft. HG: Data curation, Investigation, Writing &#x02013; original draft. BH: Methodology, Writing &#x02013; review &#x00026; editing. SH: Supervision, Validation, Writing &#x02013; review &#x00026; editing. JH: Methodology, Visualization, Writing &#x02013; review &#x00026; editing. SL: Methodology, Supervision, Writing &#x02013; review &#x00026; editing. XWu: Supervision, Validation, Writing &#x02013; review &#x00026; editing. C-SL: Methodology, Supervision, Writing &#x02013; review &#x00026; editing. XWa: Conceptualization, Project administration, Supervision, Writing &#x02013; review &#x00026; editing.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>BH was employed by Dropbox Inc.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s11">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p></sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Aneja</surname> <given-names>S.</given-names></name> <name><surname>Midoglu</surname> <given-names>C.</given-names></name> <name><surname>Dang-Nguyen</surname> <given-names>D.-T.</given-names></name> <name><surname>Khan</surname> <given-names>S. A.</given-names></name> <name><surname>Riegler</surname> <given-names>M.</given-names></name> <name><surname>Halvorsen</surname> <given-names>P.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Acm multimedia grand challenge on detecting cheapfakes</article-title>. <source>ArXiv, abs/2207.14534</source>.</mixed-citation>
</ref>
<ref id="B2">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bagozzi</surname> <given-names>B. E.</given-names></name> <name><surname>Goel</surname> <given-names>R.</given-names></name> <name><surname>Lugo-De-Fabritz</surname> <given-names>B.</given-names></name> <name><surname>Knickmeier-Cummings</surname> <given-names>K.</given-names></name> <name><surname>Balasubramanian</surname> <given-names>K.</given-names></name></person-group> (<year>2024</year>). <article-title>A framework for enhancing social media misinformation detection with topical-tactics</article-title>. <source>Dig. Threat</source>. <volume>5</volume>, <fpage>1</fpage>&#x02013;<lpage>29</lpage>. <pub-id pub-id-type="doi">10.1145/3670694</pub-id></mixed-citation>
</ref>
<ref id="B3">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bondielli</surname> <given-names>A.</given-names></name> <name><surname>Marcelloni</surname> <given-names>F.</given-names></name></person-group> (<year>2019</year>). <article-title>A survey on fake news and rumour detection techniques</article-title>. <source>Inf. Sci</source>. <volume>497</volume>, <fpage>38</fpage>&#x02013;<lpage>55</lpage>. <pub-id pub-id-type="doi">10.1016/j.ins.2019.05.035</pub-id></mixed-citation>
</ref>
<ref id="B4">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brown</surname> <given-names>T. B.</given-names></name> <name><surname>Mann</surname> <given-names>B.</given-names></name> <name><surname>Ryder</surname> <given-names>N.</given-names></name> <name><surname>Subbiah</surname> <given-names>M.</given-names></name> <name><surname>Kaplan</surname> <given-names>J.</given-names></name> <name><surname>Dhariwal</surname> <given-names>P.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Language models are few-shot learners</article-title>. <source>ArXiv, abs/2005.14165</source>.</mixed-citation>
</ref>
<ref id="B5">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Chadha</surname> <given-names>A.</given-names></name> <name><surname>Kumar</surname> <given-names>V.</given-names></name> <name><surname>Kashyap</surname> <given-names>S.</given-names></name> <name><surname>Gupta</surname> <given-names>M.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Deepfake: an overview,&#x0201D;</article-title> in <source>Proceedings of second international conference on computing, communications, and cyber-security: IC4S 2020</source> (<publisher-loc>Springer</publisher-loc>), <fpage>557</fpage>&#x02013;<lpage>566</lpage>. <pub-id pub-id-type="doi">10.1007/978-981-16-0733-2_39</pub-id></mixed-citation>
</ref>
<ref id="B6">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Zheng</surname> <given-names>P.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Hu</surname> <given-names>S.</given-names></name> <name><surname>Zhu</surname> <given-names>B.</given-names></name> <name><surname>Hu</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>&#x0201C;Harnessing the power of text-image contrastive models for automatic detection of online misinformation,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) Workshops</source>, 923&#x02013;932. <pub-id pub-id-type="doi">10.1109/CVPRW59228.2023.00099</pub-id></mixed-citation>
</ref>
<ref id="B7">
<mixed-citation publication-type="web"><person-group person-group-type="author"><name><surname>Chiang</surname> <given-names>W.-L.</given-names></name> <name><surname>Li</surname> <given-names>Z.</given-names></name> <name><surname>Lin</surname> <given-names>Z.</given-names></name> <name><surname>Sheng</surname> <given-names>Y.</given-names></name> <name><surname>Wu</surname> <given-names>Z.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality, march 2023</article-title>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://lmsys.org/blog/2023-03-30-vicuna">https://lmsys.org/blog/2023-03-30-vicuna</ext-link> <volume>3</volume>:<fpage>6</fpage>.</mixed-citation>
</ref>
<ref id="B8">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Devlin</surname> <given-names>J.</given-names></name> <name><surname>Chang</surname> <given-names>M.-W.</given-names></name> <name><surname>Lee</surname> <given-names>K.</given-names></name> <name><surname>Toutanova</surname> <given-names>K.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Bert: pre-training of deep bidirectional transformers for language understanding,&#x0201D;</article-title> in <source>Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies</source>, <fpage>4171</fpage>&#x02013;<lpage>4186</lpage>.</mixed-citation>
</ref>
<ref id="B9">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Giachanou</surname> <given-names>A.</given-names></name> <name><surname>Zhang</surname> <given-names>G.</given-names></name> <name><surname>Rosso</surname> <given-names>P.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Multimodal multi-image fake news detection,&#x0201D;</article-title> in <source>2020 IEEE 7th International Conference on Data Science and Advanced Analytics (DSAA)</source>, 647&#x02013;654. <pub-id pub-id-type="doi">10.1109/DSAA49011.2020.00091</pub-id></mixed-citation>
</ref>
<ref id="B10">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Granik</surname> <given-names>M.</given-names></name> <name><surname>Mesyura</surname> <given-names>V.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Fake news detection using naive bayes classifier,&#x0201D;</article-title> in <source>2017 IEEE First Ukraine Conference on Electrical and Computer Engineering (UKRCON)</source>, 900&#x02013;903. <pub-id pub-id-type="doi">10.1109/UKRCON.2017.8100379</pub-id></mixed-citation>
</ref>
<ref id="B11">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Guo</surname> <given-names>B.</given-names></name> <name><surname>Ding</surname> <given-names>Y.</given-names></name> <name><surname>Yao</surname> <given-names>L.</given-names></name> <name><surname>Liang</surname> <given-names>Y.</given-names></name> <name><surname>Yu</surname> <given-names>Z.</given-names></name></person-group> (<year>2019</year>). <article-title>The future of misinformation detection: new perspectives and trends</article-title>. <source>ArXiv, abs/1909.03654</source>.</mixed-citation>
</ref>
<ref id="B12">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Guo</surname> <given-names>H.</given-names></name> <name><surname>Hu</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Chang</surname> <given-names>M.-C.</given-names></name> <name><surname>Lyu</surname> <given-names>S.</given-names></name></person-group> (<year>2022a</year>). <article-title>&#x0201C;Eyes tell all: irregular pupil shapes reveal gan-generated faces,&#x0201D;</article-title> in <source>ICASSP 2022&#x02013;2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>2904</fpage>&#x02013;<lpage>2908</lpage>. <pub-id pub-id-type="doi">10.1109/ICASSP43922.2022.9746597</pub-id></mixed-citation>
</ref>
<ref id="B13">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Guo</surname> <given-names>H.</given-names></name> <name><surname>Hu</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Chang</surname> <given-names>M.-C.</given-names></name> <name><surname>Lyu</surname> <given-names>S.</given-names></name></person-group> (<year>2022b</year>). <article-title>Open-eye: an open platform to study human performance on identifying ai-synthesized faces</article-title>. <source>arXiv preprint arXiv:2205.06680</source>.</mixed-citation>
</ref>
<ref id="B14">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Guo</surname> <given-names>H.</given-names></name> <name><surname>Hu</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Chang</surname> <given-names>M.-C.</given-names></name> <name><surname>Lyu</surname> <given-names>S.</given-names></name></person-group> (<year>2022c</year>). <article-title>Robust attentive deep neural network for exposing gan-generated faces</article-title>. <source>IEEE Access</source> <volume>10</volume>, <fpage>32574</fpage>&#x02013;<lpage>32583</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2022.3157297</pub-id></mixed-citation>
</ref>
<ref id="B15">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jiang</surname> <given-names>T.</given-names></name> <name><surname>Li</surname> <given-names>J. P.</given-names></name> <name><surname>Haq</surname> <given-names>A. U.</given-names></name> <name><surname>Saboor</surname> <given-names>A.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Fake news detection using deep recurrent neural networks,&#x0201D;</article-title> in <source>2020 17th International Computer Conference on Wavelet Active Media Technology and Information Processing (ICCWAMTIP)</source>, 205&#x02013;208. <pub-id pub-id-type="doi">10.1109/ICCWAMTIP51612.2020.9317325</pub-id></mixed-citation>
</ref>
<ref id="B16">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jin</surname> <given-names>X.</given-names></name> <name><surname>Chen</surname> <given-names>P.-Y.</given-names></name> <name><surname>Hsu</surname> <given-names>C.-Y.</given-names></name> <name><surname>Yu</surname> <given-names>C.-M.</given-names></name> <name><surname>Chen</surname> <given-names>T.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Cafe: catastrophic data leakage in vertical federated learning,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source>, eds. M. Ranzato, A. Beygelzimer, Y. Dauphin, P. Liang, and J. W. Vaughan (Curran Associates, Inc.), <fpage>994</fpage>&#x02013;<lpage>1006</lpage>.</mixed-citation>
</ref>
<ref id="B17">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kaliyar</surname> <given-names>R. K.</given-names></name> <name><surname>Goswami</surname> <given-names>A.</given-names></name> <name><surname>Narang</surname> <given-names>P.</given-names></name> <name><surname>Sinha</surname> <given-names>S.</given-names></name></person-group> (<year>2020</year>). <article-title>Fndnet &#x02013; a deep convolutional neural network for fake news detection</article-title>. <source>Cogn. Syst. Res</source>. <volume>61</volume>, <fpage>32</fpage>&#x02013;<lpage>44</lpage>. <pub-id pub-id-type="doi">10.1016/j.cogsys.2019.12.005</pub-id></mixed-citation>
</ref>
<ref id="B18">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kendall</surname> <given-names>A.</given-names></name> <name><surname>Gal</surname> <given-names>Y.</given-names></name> <name><surname>Cipolla</surname> <given-names>R.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Multi-task learning using uncertainty to weigh losses for scene geometry and semantics,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source>, 7482&#x02013;7491. <pub-id pub-id-type="doi">10.1109/CVPR.2018.00781</pub-id></mixed-citation>
</ref>
<ref id="B19">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>D.</given-names></name> <name><surname>Savarese</surname> <given-names>S.</given-names></name> <name><surname>Hoi</surname> <given-names>S.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models,&#x0201D;</article-title> in <source>International Conference on Machine Learning</source> (<publisher-loc>PMLR</publisher-loc>), <fpage>19730</fpage>&#x02013;<lpage>19742</lpage>.</mixed-citation>
</ref>
<ref id="B20">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Malthouse</surname> <given-names>E. C.</given-names></name></person-group> (<year>2024</year>). <article-title>Large language model agent for fake news detection</article-title>. <source>ArXiv, abs/2405.01593</source>.</mixed-citation>
</ref>
<ref id="B21">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>P.</given-names></name> <name><surname>Huang</surname> <given-names>H.</given-names></name> <name><surname>Li</surname> <given-names>Z.</given-names></name> <name><surname>Cui</surname> <given-names>X.</given-names></name> <name><surname>Liang</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>&#x0201C;Fka-owl: advancing multimodal fake news detection through knowledge-augmented lvlms,&#x0201D;</article-title> in <source>Proceedings of the 32nd ACM International Conference on Multimedia</source>, 10154&#x02013;10163. <pub-id pub-id-type="doi">10.1145/3664647.3681089</pub-id></mixed-citation>
</ref>
<ref id="B22">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Meel</surname> <given-names>P.</given-names></name> <name><surname>Vishwakarma</surname> <given-names>D. K.</given-names></name></person-group> (<year>2020</year>). <article-title>Fake news, rumor, information pollution in social media and web: a contemporary survey of state-of-the-arts, challenges and opportunities</article-title>. <source>Expert Syst. Appl</source>. <volume>153</volume>:<fpage>112986</fpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2019.112986</pub-id></mixed-citation>
</ref>
<ref id="B23">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nakamura</surname> <given-names>K.</given-names></name> <name><surname>Levy</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>W. Y.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Fakeddit: a new multimodal benchmark dataset for fine-grained fake news detection,&#x0201D;</article-title> in <source>International Conference on Language Resources and Evaluation</source>.</mixed-citation>
</ref>
<ref id="B24">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nasir</surname> <given-names>J. A.</given-names></name> <name><surname>Khan</surname> <given-names>O. S.</given-names></name> <name><surname>Varlamis</surname> <given-names>I.</given-names></name></person-group> (<year>2021</year>). <article-title>Fake news detection: a hybrid CNN-RNN based deep learning approach</article-title>. <source>Int. J. Inf. Manag. Data Insights</source> <volume>1</volume>:<fpage>100007</fpage>. <pub-id pub-id-type="doi">10.1016/j.jjimei.2020.100007</pub-id></mixed-citation>
</ref>
<ref id="B25">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Papadopoulos</surname> <given-names>S.-I.</given-names></name> <name><surname>Koutlis</surname> <given-names>C.</given-names></name> <name><surname>Papadopoulos</surname> <given-names>S.</given-names></name> <name><surname>Petrantonakis</surname> <given-names>P. C.</given-names></name></person-group> (<year>2023</year>). <article-title>Verite: a robust benchmark for multimodal misinformation detection accounting for unimodal bias</article-title>. <source>Int. J. Multim. Inf. Retr</source>. <volume>13</volume>:<fpage>4</fpage>. <pub-id pub-id-type="doi">10.1007/s13735-023-00312-6</pub-id></mixed-citation>
</ref>
<ref id="B26">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pu</surname> <given-names>W.</given-names></name> <name><surname>Hu</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Hu</surname> <given-names>S.</given-names></name> <name><surname>Zhu</surname> <given-names>B.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Learning a deep dual-level network for robust deepfake detection</article-title>. <source>Patt. Recognit</source>. <volume>130</volume>:<fpage>108832</fpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2022.108832</pub-id></mixed-citation>
</ref>
<ref id="B27">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Qi</surname> <given-names>P.</given-names></name> <name><surname>Cao</surname> <given-names>J.</given-names></name> <name><surname>Yang</surname> <given-names>T.</given-names></name> <name><surname>Guo</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Exploiting multi-domain visual information for fake news detection,&#x0201D;</article-title> in <source>2019 IEEE International Conference on Data Mining (ICDM)</source>, 518&#x02013;527. <pub-id pub-id-type="doi">10.1109/ICDM.2019.00062</pub-id></mixed-citation>
</ref>
<ref id="B28">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Qian</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Hu</surname> <given-names>J.</given-names></name> <name><surname>Fang</surname> <given-names>Q.</given-names></name> <name><surname>Xu</surname> <given-names>C.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Hierarchical multi-modal contextual attention network for fake news detection,&#x0201D;</article-title> in <source>Proceedings of the 44th International ACM SIGIR Conference on Research and Development in Information Retrieval</source>. <pub-id pub-id-type="doi">10.1145/3404835.3462871</pub-id></mixed-citation>
</ref>
<ref id="B29">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Simonyan</surname> <given-names>K.</given-names></name> <name><surname>Zisserman</surname> <given-names>A.</given-names></name></person-group> (<year>2014</year>). <article-title>Very deep convolutional networks for large-scale image recognition</article-title>. <source>CoRR, abs/1409.1556</source>.</mixed-citation>
</ref>
<ref id="B30">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Singhal</surname> <given-names>S.</given-names></name> <name><surname>Shah</surname> <given-names>R. R.</given-names></name> <name><surname>Chakraborty</surname> <given-names>T.</given-names></name> <name><surname>Kumaraguru</surname> <given-names>P.</given-names></name> <name><surname>Satoh</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Spotfake: a multi-modal framework for fake news detection,&#x0201D;</article-title> in <source>2019 IEEE Fifth International Conference on Multimedia Big Data (BigMM), pages</source> 39&#x02013;47. <pub-id pub-id-type="doi">10.1109/BigMM.2019.00-44</pub-id></mixed-citation>
</ref>
<ref id="B31">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Su</surname> <given-names>J.</given-names></name> <name><surname>Cardie</surname> <given-names>C.</given-names></name> <name><surname>Nakov</surname> <given-names>P.</given-names></name></person-group> (<year>2023</year>). <article-title>Adapting fake news detection to the era of large language models</article-title>. <source>ArXiv, abs/2311.04917</source>. <pub-id pub-id-type="doi">10.18653/v1/2024.findings-naacl.95</pub-id></mixed-citation>
</ref>
<ref id="B32">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sudhakar</surname> <given-names>M.</given-names></name> <name><surname>Kaliyamurthie</surname> <given-names>K.</given-names></name></person-group> (<year>2023</year>). <source>Fake News Detection Approach Based on Logistic Regression in Machine Learning</source>. Cham: Springer, <fpage>55</fpage>&#x02013;<lpage>60</lpage>. <pub-id pub-id-type="doi">10.1007/978-981-19-9304-6_6</pub-id></mixed-citation>
</ref>
<ref id="B33">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Teo</surname> <given-names>T. W.</given-names></name> <name><surname>Chua</surname> <given-names>H. N.</given-names></name> <name><surname>Jasser</surname> <given-names>M. B.</given-names></name> <name><surname>Wong</surname> <given-names>R. T.</given-names></name></person-group> (<year>2024</year>). <article-title>&#x0201C;Integrating large language models and machine learning for fake news detection,&#x0201D;</article-title> in <source>2024 20th IEEE International Colloquium on Signal Processing &#x00026;Its Applications (CSPA)</source>, 102&#x02013;107. <pub-id pub-id-type="doi">10.1109/CSPA60979.2024.10525308</pub-id></mixed-citation>
</ref>
<ref id="B34">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Guo</surname> <given-names>H.</given-names></name> <name><surname>Hu</surname> <given-names>S.</given-names></name> <name><surname>Chang</surname> <given-names>M.-C.</given-names></name> <name><surname>Lyu</surname> <given-names>S.</given-names></name></person-group> (<year>2023</year>). <article-title>GAN-generated faces detection: a survey and new perspectives</article-title>. <source>arXiv preprint arXiv:2202.07145</source>.</mixed-citation>
</ref>
<ref id="B35">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Ma</surname> <given-names>F.</given-names></name> <name><surname>Jin</surname> <given-names>Z.</given-names></name> <name><surname>Yuan</surname> <given-names>Y.</given-names></name> <name><surname>Xun</surname> <given-names>G.</given-names></name> <name><surname>Jha</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>&#x0201C;Eann: event adversarial neural networks for multi-modal fake news detection,&#x0201D;</article-title> in <source>Proceedings of the 24th ACM SIGKDD International Conference on Knowledge Discovery &#x00026;Data Mining</source>. <pub-id pub-id-type="doi">10.1145/3219819.3219903</pub-id></mixed-citation>
</ref>
<ref id="B36">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>D.</given-names></name> <name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Shen</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Elhoseiny</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). <article-title>Minigpt-4: enhancing vision-language understanding with advanced large language models</article-title>. <source>ArXiv, abs/2304.10592</source>.</mixed-citation>
</ref>
<ref id="B37">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhuang</surname> <given-names>L.</given-names></name> <name><surname>Wayne</surname> <given-names>L.</given-names></name> <name><surname>Ya</surname> <given-names>S.</given-names></name> <name><surname>Jun</surname> <given-names>Z.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;A robustly optimized BERT pre-training approach with post-training,&#x0201D;</article-title> in <source>Proceedings of the 20th Chinese National Conference on Computational Linguistics</source>, eds. S. Li, M. Sun, Y. Liu, H. Wu, K., W. Che, S. He, and G. Rao (Huhhot, China: Chinese Information Processing Society of China), <fpage>1218</fpage>&#x02013;<lpage>1227</lpage>.</mixed-citation>
</ref>
</ref-list>
<fn-group>
<fn fn-type="custom" custom-type="edited-by" id="fn0001">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1999091/overview">Jiaqi Gong</ext-link>, University of Alabama, United States</p>
</fn>
<fn fn-type="custom" custom-type="reviewed-by" id="fn0002">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2258357/overview">Antonio Sarasa-Cabezuelo</ext-link>, Complutense University of Madrid, Spain</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3171390/overview">Xiaoming Guo</ext-link>, University of Alabama System, United States</p>
</fn>
</fn-group>
</back>
</article>