<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="brief-report">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2024.1361483</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Brief Research Report</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Streamlining event extraction with a simplified annotation framework</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Saetia</surname> <given-names>Chanatip</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Thonglong</surname> <given-names>Areeya</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Amornchaiteera</surname> <given-names>Thanpitcha</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Chalothorn</surname> <given-names>Tawunrat</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Taerungruang</surname> <given-names>Supawat</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2658789/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Buabthong</surname> <given-names>Pakpoom</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2597473/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Kasikorn Labs, Kasikorn Business-Technology Group</institution>, <addr-line>Nonthaburi</addr-line>, <country>Thailand</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Thai, Faculty of Humanities, Chiangmai University</institution>, <addr-line>Chiang Mai</addr-line>, <country>Thailand</country></aff>
<aff id="aff3"><sup>3</sup><institution>Faculty of Science and Technology, Nakhon Ratchasima Rajabhat University</institution>, <addr-line>Nakhon Ratchasima</addr-line>, <country>Thailand</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Jennifer D&#x00027;Souza, Technische Informationsbibliothek (TIB), Germany</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Hamed Babaei Giglou, Technische Informationsbibliothek (TIB), Germany</p>
<p>Azanzi Jiomekong, University of Yaounde I, Cameroon</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Pakpoom Buabthong <email>pakpoom.b&#x00040;nrru.ac.th</email></corresp></author-notes>
<pub-date pub-type="epub">
<day>29</day>
<month>04</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>7</volume>
<elocation-id>1361483</elocation-id>
<history>
<date date-type="received">
<day>26</day>
<month>12</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>08</day>
<month>04</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2024 Saetia, Thonglong, Amornchaiteera, Chalothorn, Taerungruang and Buabthong.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Saetia, Thonglong, Amornchaiteera, Chalothorn, Taerungruang and Buabthong</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Event extraction, grounded in semantic relationships, can serve as a simplified relation extraction. In this study, we propose an efficient open-domain event annotation framework tailored for subsequent information extraction, with a specific focus on its applicability to low-resource languages. The proposed event annotation method, which is based on event semantic elements, demonstrates substantial time-efficiency gains over traditional Universal Dependencies (UD) tagging. We show how language-specific pretraining outperforms multilingual counterparts in entity and relation extraction tasks and emphasize the importance of task- and language-specific fine-tuning for optimal model performance. Furthermore, we demonstrate the improvement of model performance upon integrating UD information during pre-training, achieving the F1 score of 71.16 and 60.43% for entity and relation extraction respectively. In addition, we showcase the usage of our extracted event graph for improving node classification in a retail banking domain. This work provides valuable guidance on improving information extraction and outlines a methodology for developing training datasets, particularly for low-resource languages.</p></abstract>
<kwd-group>
<kwd>event extraction</kwd>
<kwd>annotation guideline</kwd>
<kwd>Universal Dependencies</kwd>
<kwd>generative model</kwd>
<kwd>event graph</kwd>
</kwd-group>
<counts>
<fig-count count="1"/>
<table-count count="3"/>
<equation-count count="0"/>
<ref-count count="79"/>
<page-count count="9"/>
<word-count count="7794"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Natural Language Processing</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>The advent of large language models (LLMs) has enabled significant progress in the field of natural language processing (NLP) and has helped provide promising results for various tasks (Brown et al., <xref ref-type="bibr" rid="B7">2020</xref>). Many types of LLM have been proposed to solve both language-specific and domain-specific tasks (Lewis et al., <xref ref-type="bibr" rid="B34">2019</xref>; Chung et al., <xref ref-type="bibr" rid="B14">2022</xref>; Touvron et al., <xref ref-type="bibr" rid="B64">2023</xref>). However, LLMs primarily favored well-resourced language with large updated training corpora, which may lead to hallucination problems, especially in lower-resource languages, in which the training corpora are not abundantly available (Ji et al., <xref ref-type="bibr" rid="B28">2023</xref>). Extracting knowledge from these low-resource languages is not only beneficial as it helps include more available data. It could also provide deeper insight into the model&#x00027;s behavior across linguistic variations. To mitigate the hallucination problem, researchers have explored augmenting LLMs with external structured data sources, such as knowledge graphs (Guu et al., <xref ref-type="bibr" rid="B22">2020</xref>; Asai et al., <xref ref-type="bibr" rid="B3">2021</xref>; Mialon et al., <xref ref-type="bibr" rid="B50">2023</xref>). Integrating structured information graphs with the LMs has been one of the common approaches (Yao et al., <xref ref-type="bibr" rid="B77">2019</xref>; Kang et al., <xref ref-type="bibr" rid="B29">2022</xref>), as graphs can be constructed in a domain-specific fashion, such as finance (Yang et al., <xref ref-type="bibr" rid="B75">2018</xref>; Elhammadi et al., <xref ref-type="bibr" rid="B18">2020</xref>).</p>
<p>Event graphs, which store event information from unstructured plain texts that describe &#x0201C;who, when, where, what, why&#x0201D; and &#x0201C;how&#x0201D; of the action, can provide a simplified version of a more generalized knowledge graph (Xiang and Wang, <xref ref-type="bibr" rid="B73">2019</xref>; Li et al., <xref ref-type="bibr" rid="B38">2022</xref>). Focusing on event extraction is particularly promising for enhancing NLP in low-resource settings because it involves parsing relationships within the narrow scope of particular events, thus requiring less extensive linguistic understanding for the model. Although close-domain event extraction, which follows specific domain schema, may provide better results in downstream retrieval tasks (Chambers et al., <xref ref-type="bibr" rid="B9">2014</xref>; Bj&#x000F6;rne and Salakoski, <xref ref-type="bibr" rid="B5">2018</xref>; Han et al., <xref ref-type="bibr" rid="B23">2018</xref>), this specialization often results in complex annotation systems that can be cumbersome and domain-restrictive, especially for low-resource languages. Moreover, while the use of additional syntactic information for extraction tasks has been studied in English (Fader et al., <xref ref-type="bibr" rid="B20">2011</xref>; Wang C. et al., <xref ref-type="bibr" rid="B69">2023</xref>; Wang Z. et al., <xref ref-type="bibr" rid="B70">2023</xref>), it remains under-explored in low-resource languages.</p>
<p>In this work, we propose a methodology that streamlines the process for open-domain event extraction for corporate documents written in Thai and demonstrates its utility in a downstream task. Our guideline aims to make structured information extraction more accessible, by reducing the complexity of the annotation process. We also utilize Universal Dependencies [UD; Nivre et al., <xref ref-type="bibr" rid="B53">2016</xref>] during the pre-training step to help the extraction model better understand the structural information of the sentences.</p>
<p>The main contributions of this work are as follows:</p>
<list list-type="bullet">
<list-item><p><bold>Annotation framework:</bold> We offer a simplified annotation guideline that streamlines the event extraction process and presents a comparative analysis with the traditional Universal Dependency (UD) framework.</p></list-item>
<list-item><p><bold>Event extraction models:</bold> We explore the impact of language-specific and task-specific pre-training as well as the incorporation of UD on the improvement of the overall extraction performance.</p></list-item>
<list-item><p><bold>Applications:</bold> We demonstrate that the extracted event graph can be utilized to improve a downstream task, namely, node classification in a retail banking domain.</p></list-item>
</list>
<p>The rest of this paper is organized as follows. Section 2 analyzes previous work. Section 3 describes our methodology. Section 4 reports on our experiments. Section 5 provides a discussion of the results. Section 6 elaborates on the application of the event graphs. Section 7 concludes with a summary. By simplifying the initial extraction process, our method could allow for a more straightforward transition into an extraction task for other types of relations, such as, part-of or causal relations, which often require a deeper understanding of the interconnectedness of entities beyond their basic semantic relationships.</p></sec>
<sec id="s2">
<title>2 Related work</title>
<p>In this section, the background of the paper is explained along with literature reviews, outlining the previous work on event extraction, and Universal Dependencies. First, the definition and prior works of event extraction are explained. Second, the Universal Dependencies are described including the definition and its advantages.</p>
<sec>
<title>2.1 Event extraction</title>
<p>Event extraction typically aims to extract event attributes from a raw, answering the 5W1H (who, what, when, where, why, and how) questions (Xiang and Wang, <xref ref-type="bibr" rid="B73">2019</xref>). In earlier work, event extraction is considered a sequence labeling-based task (Gupta and Manning, <xref ref-type="bibr" rid="B21">2014</xref>; Chen et al., <xref ref-type="bibr" rid="B11">2020</xref>). The event trigger and its arguments are extracted as a span of words with an inside-outside-beginning (BIO) tagging system (Li et al., <xref ref-type="bibr" rid="B38">2022</xref>). However, multiple events may be found in a given sentence, thus later necessitating the classification of the relation between each argument with its trigger.</p>
<p>The event extraction task can generally be categorized into two groups: close domain and open domain (Xiang and Wang, <xref ref-type="bibr" rid="B73">2019</xref>; Liu et al., <xref ref-type="bibr" rid="B39">2021</xref>, <xref ref-type="bibr" rid="B40">2023</xref>). The close-domain extraction aims to extract a pre-defined structure based on supervised datasets. Most approaches first identify the event trigger, followed by its corresponding attributes (Huang et al., <xref ref-type="bibr" rid="B27">2017</xref>; Xiang and Wang, <xref ref-type="bibr" rid="B73">2019</xref>). Each event attributed is connected to the trigger with a pre-defined relation.</p>
<p>Various methods were proposed to address close-domain extraction (Chen et al., <xref ref-type="bibr" rid="B12">2015a</xref>; Huang et al., <xref ref-type="bibr" rid="B27">2017</xref>; Li et al., <xref ref-type="bibr" rid="B37">2020</xref>). Some treat the event extraction as a sequence of sub-tasks: trigger identification, Trigger classification, argument identification, and argument role classification (Chen et al., <xref ref-type="bibr" rid="B13">2015b</xref>; Yang et al., <xref ref-type="bibr" rid="B76">2019</xref>; Li et al., <xref ref-type="bibr" rid="B38">2022</xref>). However, this technique could lead to error propagation during the process (Li et al., <xref ref-type="bibr" rid="B35">2019</xref>; Zhang et al., <xref ref-type="bibr" rid="B79">2019</xref>). To minimize this error propagation, joint-trained models were proposed (Hsu et al., <xref ref-type="bibr" rid="B25">2021</xref>; Lu et al., <xref ref-type="bibr" rid="B46">2021</xref>). Many approaches adopt deep learning model architecture to train an end-to-end event extraction (Nguyen and Nguyen, <xref ref-type="bibr" rid="B52">2019</xref>; Wadden et al., <xref ref-type="bibr" rid="B67">2019</xref>). Recently, conditional generations from language models yield promising accuracy among many NLP tasks. Such models have been adopted for event extraction, achieving state-of-the-art accuracy over the complex classification models (Hsu et al., <xref ref-type="bibr" rid="B25">2021</xref>; Lu et al., <xref ref-type="bibr" rid="B46">2021</xref>). Nevertheless, the learning for these deep learning model approaches is supervised, necessitating a large amount of training data, which is not practical for low-resource languages.</p>
<p>Although the accuracy of the close-domain models is promising, most datasets are still limited to specific domains like medical data, historical documents, or specific types of news (Vanegas et al., <xref ref-type="bibr" rid="B65">2015</xref>; Bj&#x000F6;rne and Salakoski, <xref ref-type="bibr" rid="B5">2018</xref>; Han et al., <xref ref-type="bibr" rid="B23">2018</xref>). Thus, to extract a generic event from more generalized corpora, open-domain event extraction was developed (Chau et al., <xref ref-type="bibr" rid="B10">2019</xref>; Liu et al., <xref ref-type="bibr" rid="B41">2019</xref>). The early model considers the headline phrase as an event and disambiguates the events using Wordnet (Miller, <xref ref-type="bibr" rid="B51">1995</xref>) and word sense disambiguation (Chau et al., <xref ref-type="bibr" rid="B10">2019</xref>). This method leads to suboptimal performance as the arguments of an event are not necessarily positioned next to an event trigger keyword. To address this limitation, another model utilizes an unsupervised method using a neural latent variable model to extract an event (Liu et al., <xref ref-type="bibr" rid="B41">2019</xref>). However, because of its unsupervised architecture, this method is not controllable and can extract an inaccurate event entity.</p></sec>
<sec>
<title>2.2 Low-resource event extraction</title>
<p>Similar to other tasks under a low-resource setting, the development of event extraction for low-resource languages generally focuses on methods that require less amount of training data. Zero-shot learning is one of the most common approaches to help the model perform tasks without additional training samples. Previous work on zero-shot event extraction has explored the use of representation in other latent spaces such as semi-Markov conditional random fields (Lu and Roth, <xref ref-type="bibr" rid="B45">2012</xref>), Abstract Meaning Representation (AMR; Huang et al., <xref ref-type="bibr" rid="B26">2018</xref>), pre-defined ontological structure (Zhang et al., <xref ref-type="bibr" rid="B78">2021</xref>). Alternatively, event extraction tasks may be formulated as different tasks such as question-answering (Lyu et al., <xref ref-type="bibr" rid="B48">2021</xref>). However, these techniques necessitate that proficient models already exist in the target language.</p>
<p>On the other hand, few-shot learning can be utilized to minimize the amount of new training data that is specific to the extraction tasks, while improving the overall performance of the models. Early models use a prototypical network to classify the extracted token (Snell et al., <xref ref-type="bibr" rid="B58">2017</xref>; Lai and Nguyen, <xref ref-type="bibr" rid="B33">2019</xref>), or minimize the supervised training data by providing the trigger terms in the annotation guideline as seeds for each event type (Bronstein et al., <xref ref-type="bibr" rid="B6">2015</xref>). More recent work addresses the issue of low sample diversity by introducing Adaptive Knowledge-Enhanced Bayesian Meta Learning (AKE-BML) that uses a prior knowledge distribution to generate the posterior distribution for each event type (Shen et al., <xref ref-type="bibr" rid="B57">2021</xref>). Techniques used in a few-shot setting typically work well when there exists a known distribution within a given task followed by model refinement through additional examples in the target tasks. For example, in Thai, we can pre-train the model with a syntactic structure such as UD, then fine-tune the model with a small number of labels for event extraction.</p>
<p>Furthermore, cross-lingual transfer may be employed when both languages have well-established parallel corpus. Recent methods have proposed transferring the entire universal structures across languages (Li et al., <xref ref-type="bibr" rid="B36">2016</xref>; Subburathinam et al., <xref ref-type="bibr" rid="B61">2019</xref>; Lou et al., <xref ref-type="bibr" rid="B43">2022</xref>), or leveraging multilingual embedding when training the extraction model (M&#x00027;hamdi et al., <xref ref-type="bibr" rid="B49">2019</xref>). However, the cross-lingual approach typically requires extensive lexical mapping which may not be suitable for this initial stage of the development.</p></sec>
<sec>
<title>2.3 Relation extraction</title>
<p>In addition to models specific to event extraction, other relation extraction models may be utilized. End-to-end deep learning models have been proposed to concurrently extract entities and relation (Bekoulis et al., <xref ref-type="bibr" rid="B4">2018</xref>; Eberts and Ulges, <xref ref-type="bibr" rid="B17">2019</xref>; Hang et al., <xref ref-type="bibr" rid="B24">2021</xref>). SpERT (Eberts and Ulges, <xref ref-type="bibr" rid="B17">2019</xref>), in particular, has shown promising results on both entity and relation extraction evaluated over the SciERC dataset (Luan et al., <xref ref-type="bibr" rid="B47">2018</xref>).</p>
<p>Moreover, generative pre-trained language models have been reported to achieve high performance on many NLP tasks (Brown et al., <xref ref-type="bibr" rid="B7">2020</xref>; Touvron et al., <xref ref-type="bibr" rid="B64">2023</xref>). Structured prediction using generative LMs, in particular, has recently attracted interest, due to their flexibility and applicability to new datasets. Most models are trained to generate structured output for named entities recognition or relation extraction from unstructured texts (Eberts and Ulges, <xref ref-type="bibr" rid="B17">2019</xref>; Lu et al., <xref ref-type="bibr" rid="B46">2021</xref>; Paolini et al., <xref ref-type="bibr" rid="B55">2021</xref>). DeepStruct (Wang C. et al., <xref ref-type="bibr" rid="B69">2023</xref>), for example, offers state-of-the-art performance when predicting the triplet from various domains, namely T-REx (Elsahar et al., <xref ref-type="bibr" rid="B19">2018</xref>), TEKGEN, KELM (Agarwal et al., <xref ref-type="bibr" rid="B1">2021</xref>), WebNLG (Colin et al., <xref ref-type="bibr" rid="B15">2016</xref>), and ConceptNet (Speer et al., <xref ref-type="bibr" rid="B59">2017</xref>).</p>
<p>Nevertheless, the models may not perform well in other languages that are not primarily present in the pre-training dataset. Other syntactic or semantic information, such as Universal Dependencies (UD) may assist in cross-lingual transfer of the extraction capabilities.</p></sec>
<sec>
<title>2.4 Universal Dependencies</title>
<p>Universal Dependencies (UD; Nivre et al., <xref ref-type="bibr" rid="B53">2016</xref>) is a cross-language framework that allows for consistency in the annotation of syntactic grammatical structure (parts of speech, morphological features, and syntactic dependencies). Given this UD, a reliable graph can be created to represent the syntactic structure of an arbitrary text. Some event extraction models have been reported to benefit from the incorporation of UD (Bj&#x000F6;rne and Salakoski, <xref ref-type="bibr" rid="B5">2018</xref>; Chau et al., <xref ref-type="bibr" rid="B10">2019</xref>). Unsupervised techniques can extract phrases and their relation from the UD graphs (Chau et al., <xref ref-type="bibr" rid="B10">2019</xref>). Other work used the output of the UD as a graph feature along with a graph neural network to improve an event extraction model (Liu et al., <xref ref-type="bibr" rid="B42">2018</xref>; Ahmad et al., <xref ref-type="bibr" rid="B2">2021</xref>). Nevertheless, developing extraction models that rely too heavily on UD may pose similar limitations to those with languages that have low annotated training data, since the models may learn to capture only the explicit syntactic relationship and not the generalized semantic structure of the sentences.</p></sec></sec>
<sec sec-type="methods" id="s3">
<title>3 Methodology</title>
<p>This section outlines the annotation process and the event extraction models used in this work.</p>
<sec>
<title>3.1 Annotation framework</title>
<p>Frameworks for annotating text typically have two distinct aspects: (1) the practical means of how to annotate, and (2) the rules governing the annotation process (Pyysalo et al., <xref ref-type="bibr" rid="B56">2012</xref>; Stenetorp et al., <xref ref-type="bibr" rid="B60">2012</xref>; Cassidy et al., <xref ref-type="bibr" rid="B8">2014</xref>). For (1), in this work, we configured INCEpTION (Klie et al., <xref ref-type="bibr" rid="B32">2018</xref>) for entity and relation tagging. For (2), the complete annotation guideline is provided in the <xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>, while the abbreviated version, along with the design reasoning, is presented as follows.</p>
<p>Briefly, instead of the traditional event annotation where the trigger verb is identified first, the events are tagged based on 5W1H questions. The annotation guideline proposed two-stage tagging, which first labels entity spans and then links the relations among them. An example of a fully annotated sentence is shown below.</p>
<p><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1361483-i0001.tif"/></p>
<p>Entities, which are graph nodes of an event graph, are extracted as triggers and their corresponding arguments, represented as word spans. These entity spans are categorized into seven types to include the semantic meaning of an entity. One of the types is denoted as <italic>Action</italic>, which is similar to the trigger of an event. Other types are the semantic type of the argument, like <italic>Person, Object</italic>, and <italic>Location</italic>.</p>
<p>After getting entity spans, the subsequent step is establishing and classifying the relations among the spans. The classified relation types are designed to primarily address WH questions, which are <italic>what, who, when</italic>, and <italic>where</italic>. The <italic>how</italic> and <italic>why</italic> are not included, since the phrase that describes these two relations can be highly subjective depending on the annotator. Nevertheless, we also include additional relations, namely, same-unit, benefit, and value, in the guideline as these relations are not semantically ambiguous and can be potentially useful for downstream information extraction tasks.</p></sec>
<sec>
<title>3.2 Event extraction models</title>
<p>Two candidate models are selected for the event extraction task based on their inference settings: generative and span-based classification.</p>
<p>Span-based joint entity and relation extraction. The two models, SpERT (Eberts and Ulges, <xref ref-type="bibr" rid="B17">2019</xref>) for span-based classification and DeepStruct (Wang C. et al., <xref ref-type="bibr" rid="B69">2023</xref>) for the generative approach, were selected based on their demonstrated state-of-the-art performance in their respective tasks. SpERT has shows superior performance in span-based classification tasks, benchmarked on CoNLL-2003 (Tjong Kim Sang and De Meulder, <xref ref-type="bibr" rid="B63">2003</xref>). Similarly Deepstruct has exhibited strong performance using generative approach on ACE-2005 (Walker and Consortium, <xref ref-type="bibr" rid="B68">2005</xref>) corpus due to their superior performance in their respective tasks.</p>
<sec>
<title>3.2.1 Span-based classification model</title>
<p>For the baseline model, SpERT is used to represent a relatively more straightforward approach to the event extract task. In this approach, the model first recognizes the spans of the token of interest (entity extraction), then, with each pair of spans, learns to classify the relation types (relation extraction). Nevertheless, both entity extraction and relation extraction are trained jointly.</p></sec>
<sec>
<title>3.2.2 Generative model</title>
<p>To study the effect of incorporating UD structure into the model, a separate model based on DeepStruct is used. The model is trained in a generative setting using a short prompt and the text of interest as the input, with the event triplets as the output (shown in <xref ref-type="fig" rid="F1">Figure 1</xref>) In the UD pre-training, two tasks are trained jointly but with different prompts: part-of-speech (POS) tagging, and dependency (DEP) tagging. In contrast to the previous work where the extracted triplets are only constrained to a few important relations, each word in the input sentence of our approach will result in its own POS and DEP triplets. Note that in this generative setting, both entity extraction and relation extraction are inferred simultaneously from the model.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>An schematic showing input and output of each generative task.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1361483-g0001.tif"/>
</fig>


<p>Since the original model is pre-trained only with the English dataset, herein we pre-trained the model on our Thai dataset (UD). After the pre-training process, the model is fine-tuned on the annotated Thai event dataset. Similar to the pre-training stage, the three tasks are trained jointly using different prompts and outputs.</p></sec></sec></sec>
<sec id="s4">
<title>4 Experiments and results</title>
<p>In this section, we compare the annotation time between event annotation using our proposed guideline and the traditional UD annotation. The annotated data was then used in a comparative study between different approaches to event extraction tasks.</p>
<sec>
<title>4.1 Time for annotation</title>
<p>Our proposed guideline was used to annotate news articles and internal corporate documents written in Thai. To measure the time for annotation, two annotators were tasked to label the documents according to our guidelines as well as the standard Thai UD annotation for 1 month. Afterward, the number of annotated sentences for each task was divided to calculate the daily average from both annotators was averaged per day and divided by the number of days in that month. The statistics of the resulting annotated data are shown in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>The data statistics of the annotated dataset of event extraction.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Statistic</bold></th>
<th valign="top" align="left"><bold>Train</bold></th>
<th valign="top" align="left"><bold>Validation</bold></th>
<th valign="top" align="left"><bold>Test</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">&#x00023; sentences</td>
<td valign="top" align="left">13,566</td>
<td valign="top" align="left">2,000</td>
<td valign="top" align="left">2,000</td>
</tr> <tr>
<td valign="top" align="left">&#x00023; entities</td>
<td valign="top" align="left">15,306</td>
<td valign="top" align="left">2,173</td>
<td valign="top" align="left">2,073</td>
</tr> <tr>
<td valign="top" align="left">&#x00023; relations</td>
<td valign="top" align="left">9,117</td>
<td valign="top" align="left">1,295</td>
<td valign="top" align="left">1,223</td>
</tr></tbody>
</table>
</table-wrap>


<p>The number of sentences annotated using our event extraction guideline compared to using the typical UD guideline are 292.77 and 19.2 sentences per day, respectively, indicating &#x0007E;10 times faster annotation speed.</p></sec>
<sec>
<title>4.2 Event extraction model</title>
<p>The annotated event dataset was used to evaluate event extraction models described in this section. First, the dataset is split into a train, validation, and test dataset with ratios of 0.78, 0.11, and 0.11, respectively (we allocated 2,000 sentences each to the validation and test split and used the remaining for the training). To evaluate the model, the micro-average F1 score, calculated separately between the entities F1 score and the relation F1 score (Eberts and Ulges, <xref ref-type="bibr" rid="B17">2019</xref>), is used. Models based on SpERT (Eberts and Ulges, <xref ref-type="bibr" rid="B17">2019</xref>) and DeepStruct (Wang C. et al., <xref ref-type="bibr" rid="B69">2023</xref>) are employed to compare the performance between a span-based classification model and a generative model. To study the effect of the language-specific pre-training, a multilingual BERT (Devlin et al., <xref ref-type="bibr" rid="B16">2019</xref>) and a Thai-specific WangchanBERTa (Lowphansirikul et al., <xref ref-type="bibr" rid="B44">2021</xref>) are used in the span-based model. Lastly, in the generative settings, the pre-training model with mT5 (Xue et al., <xref ref-type="bibr" rid="B74">2021</xref>) is compared to pre-training with our Thai UD dataset. All models are fine-tuned with the annotated event training set for the event extraction task.</p>
<p><xref ref-type="table" rid="T2">Table 2</xref> shows the micro-average F1 score for entity and relation extraction. For the span-based model, using language-specific pre-training substantially outperforms the multilingual one for both entity (66.97 vs. 44.58) and relation extraction (59.20 vs. 33.86). In our settings, generative models yield better results than the span-based ones. Notably, for entity extraction, the generative model trained with the multilingual pre-training can still outperform the language-specific span-based model (68.87 vs. 66.97). Finally, the best result in both entity and relation extraction is achieved when using language-specific UD pre-training (71.16 for entity extraction and 60.43 for relation extraction).</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>The result of entity and relation extraction for event extraction of each model.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="left"><bold>P (%)</bold></th>
<th valign="top" align="left"><bold>R (%)</bold></th>
<th valign="top" align="left"><bold>F1 (%)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Entity</td>
<td valign="top" align="left">SpERT (multilingual BERT)</td>
<td valign="top" align="left">37.05</td>
<td valign="top" align="left">55.93</td>
<td valign="top" align="left">44.58</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">SpERT (Wangchanberta)</td>
<td valign="top" align="left">68.27</td>
<td valign="top" align="left">65.71</td>
<td valign="top" align="left">66.97</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">DeepStruct (mT5)</td>
<td valign="top" align="left">78.22</td>
<td valign="top" align="left">61.52</td>
<td valign="top" align="left">68.87</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">DeepStruct (UD)</td>
<td valign="top" align="left">78.96</td>
<td valign="top" align="left">64.76</td>
<td valign="top" align="left">71.16</td>
</tr> <tr>
<td valign="top" align="left">Relation</td>
<td valign="top" align="left">SpERT (multilingual BERT)</td>
<td valign="top" align="left">25.44</td>
<td valign="top" align="left">50.58</td>
<td valign="top" align="left">33.86</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">SpERT (Wangchanberta)</td>
<td valign="top" align="left">60.21</td>
<td valign="top" align="left">58.22</td>
<td valign="top" align="left">59.20</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">DeepStruct (mT5)</td>
<td valign="top" align="left">50.69</td>
<td valign="top" align="left">65.49</td>
<td valign="top" align="left">57.15</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">DeepStruct (UD)</td>
<td valign="top" align="left">53.56</td>
<td valign="top" align="left">69.33</td>
<td valign="top" align="left">60.43</td>
</tr></tbody>
</table>
</table-wrap>

</sec></sec>
<sec sec-type="discussion" id="s5">
<title>5 Discussion</title>
<p>Compared to the baseline UD tagging, event annotation following our guideline is substantially faster. The decrease can be attributed not only to the fewer number of relations but also to the less complex annotation scheme that the annotators need to process. Annotating using our proposed guideline mostly follows the semantic structure of the sentence, eliminating the need to recognize minor syntactic relations like &#x0201C;case,&#x0201D; or &#x0201C;disclose.&#x0201D; The more complicated relations between clauses like &#x0201C;acl,&#x0201D; &#x0201C;advcl,&#x0201D; &#x0201C;csubj,&#x0201D; or &#x0201C;xcomp&#x0201D; are also omitted. In addition, event annotation treats multiple-word phrases as single units, eliminating the need to understand the intraterm connection. As a result, when developing the data for structural information extraction models, starting from semantic relations similar to the proposed event extraction could be more practical and time-efficient, especially for languages with no pre-existing structural training data.</p>
<p>From the subsequent span-based classification result, the model using language-specific pretraining outperforms the multilingual one in both entity and relation extraction, likely attributed to both language-specific and task-specific fine-tuning. Previous work has reported that using multilingual BERT performs substantially worse for low-resource languages, like Thai, as it does not benefit from cross-lingual transfer (Wu and Dredze, <xref ref-type="bibr" rid="B71">2020</xref>) and shows that monolingual BERT-based models perform even worse for NER, POS, DEP tagging. In our case, we show that fine-tuning using task- and language-specific data offers an option to improve upon the monolingual BERT-based models.</p>
<p>When comparing the models in different settings, although the generative model with multilingual pretraining outperforms most of the span-based ones, it still lags behind the monolingual SpERT on the relation extraction task. This discrepancy is likely because the entity recognition task can benefit from the encoder-decoder architecture used in this work. A similar observation has also been previously reported (Wu et al., <xref ref-type="bibr" rid="B72">2023</xref>). Nevertheless, specific downstream tasks must be taken into account when selecting candidate baseline models, as other types, such as masked LMs, could be computationally cheaper for domain-specific training.</p>
<p>In contrast to entity extraction, the relation extraction task could benefit more from the span-based two-step classification architecture. While SpERT inherently approaches relation extraction as a direct classification task, the generative-based method necessitates the simultaneous learning of relation generation with the identification of the entities of interest.</p>
<p>Lastly, when UD is included during the pre- training stage, the generative model outperforms in both tasks. Using UD information allows the model to learn the syntactic structure of the language, potentially aiding in the semantic inference of the subsequent relation extraction.</p>
<p>This result motivates the use of UD in conjunction with a more simplified event annotation framework when developing models for structure extraction, especially for low-resource languages. Although UD annotation is substantially more time-consuming, our work shows that including such information is likely beneficial to the subsequent semantic-related tasks.</p></sec>
<sec id="s6">
<title>6 Applications of event graphs</title>
<p>After obtaining the list of event attributes from the event extraction model, these sets of structured event information can be adopted to enhance other downstream tasks. In this section, we demonstrate the application of the extracted event graph to improve node classification in the retail banking product domain. Additionally, we explore the potential of transforming our event graph into a more generic knowledge graph where the types of relations are not constrained to only those present in our event annotation guideline.</p>
<p>The event graph in this experiment was constructed from the list of event triplets extracted using the UD-pretrained model from a set of 6,024 internal documents written in Thai, describing the details of financial products and services. This results in 69,801 nodes and 168,964 relations. Out of the total entity nodes, 500 nodes were selected and labeled into one of the 15 categories: &#x0201C;Process,&#x0201D; &#x0201C;Debit,&#x0201D; &#x0201C;Credit,&#x0201D; &#x0201C;Loan,&#x0201D; &#x0201C;Service,&#x0201D; &#x0201C;Promotion,&#x0201D; &#x0201C;System,&#x0201D; &#x0201C;Right,&#x0201D; &#x0201C;Fee,&#x0201D; &#x0201C;Insurance,&#x0201D; &#x0201C;Document,&#x0201D; &#x0201C;Contact,&#x0201D; &#x0201C;Account,&#x0201D; &#x0201C;Statement,&#x0201D; and &#x0201C;RewardPoint.&#x0201D; These nodes were selected such that the resulting 500-node sub-graphs were sufficiently connected (no disconnected graphs), and the numbers of each label were balanced. The averaged F1-score of 5-fold cross-validation of these 500-node sub-graph was then used to assess the performance of the model.</p>
<p>In the baseline model, only the text embedding derived from a pre-trained Thai language model, Wangchanberta (Lowphansirikul et al., <xref ref-type="bibr" rid="B44">2021</xref>), was used. For our model, the node embedding derived from the event graph using Hash-GNN (Tan et al., <xref ref-type="bibr" rid="B62">2020</xref>) was concatenated with the original text embedding as an additional feature.</p>
<p><xref ref-type="table" rid="T3">Table 3</xref> shows the averaged F1 scores of the model using text embedding or text&#x0002B;node embedding as features. The result shows an &#x0007E;2 percentage point improvement (77.71% from 75.87%) when the model uses node embedding in conjunction with text embedding. This improvement underscores the significance of the relational information provided by our event graph using the simple Hash-GNN. To achieve further improvement, one could employ more advanced (though computationally more expensive) node embedding techniques, namely, GCN (Kipf and Welling, <xref ref-type="bibr" rid="B31">2017</xref>) or GAN (Veli&#x0010D;kovi&#x00107; et al., <xref ref-type="bibr" rid="B66">2018</xref>). In addition to the improved performance, our node classification approach adaptable to other domains and can assist organizations in processing large textual data. A similar technique could be employed to categorize entity names present in internal documents, by labeling small subset samples and then using a classification model with the extracted event graph to incorporate contextual information.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>The comparison between models with and without node embedding as a feature.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="left"><bold>F1-macro (%)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">W/o node embedding</td>
<td valign="top" align="left">75.87</td>
</tr> <tr>
<td valign="top" align="left">W/ node embedding</td>
<td valign="top" align="left">77.71</td>
</tr></tbody>
</table>
</table-wrap>


<p>Moreover, our extracted event graph can also be merged and reformatted to construct a more generic knowledge graph. Briefly, the procedure involves finding a pair of triplets such that the head entity of one pair is the same as the tail entity of the other pair. For example, the sentence &#x0201C;A criminal, previously exorenated, stole a car&#x0201D; would be converted into {subj, rel, obj} = {A criminal, stole, a car}. By merging the triplets afterward, the model is allowed to be trained under the constraint of recognizing only seven predefined relation types, yet allowing the extracted triplets to be rearranged to cover more generalized relations. Such a generalized knowledge graph can then be applied to assist in other domain-specific or language-specific information retrieval tasks, such as question answering on knowledge graphs (KGQA; Khongcharoen et al., <xref ref-type="bibr" rid="B30">2022</xref>), or KG-enhanced LLMs (Pan et al., <xref ref-type="bibr" rid="B54">2023</xref>).</p></sec>
<sec sec-type="conclusions" id="s7">
<title>7 Conclusion</title>
<p>In this paper, we introduced a streamlined event annotation framework that allows for substantially faster labeling over the baseline UD tagging. We propose that initiating the development of data for structural information extraction models with simple semantic relations, akin to event extraction, proves more practical, particularly for languages with no pre-existing structural training data.</p>
<p>Language-specific pretraining helps achieve better performance over the multilingual counterparts in both entity and relation extraction tasks. Notably, we underscored the importance of fine-tuning using task- and language-specific data to improve upon monolingual BERT-based models.</p>
<p>Under different settings, while the generative model with multilingual pretraining generally performs well, the span-based two-step classification architecture of SpERT shows a particular advantage for relation extraction tasks. The integration of UD information during the pre-training stage further improved the performance in both tasks, indicating a potential synergistic relationship between syntactic structure understanding and subsequent semantic inference.</p>
<p>Moreover, we leveraged the structured event information obtained from the event extraction model to improve node classification in the retail banking product domain. We also proposed a simple method for converting our event graph into a more generic knowledge graph that expands beyond our event relation types.</p>
<p>In conclusion, our research underscores the value of semantic-based event extraction, language-specific pretraining, and the integration of syntactic structure understanding through UD for improved performance in structural information extraction tasks. The methods we propose are not only efficient but also versatile, with potential applications in other domains, especially for developing similar structural training data for low-resource languages.</p></sec>
<sec sec-type="data-availability" id="s8">
<title>Data availability statement</title>
<p>The data supporting the conclusions of this article will be made available by the authors, upon reasonable request.</p></sec>
<sec sec-type="author-contributions" id="s9">
<title>Author contributions</title>
<p>CS: Formal analysis, Investigation, Methodology, Validation, Writing &#x02013; original draft. AT: Data curation, Formal analysis, Writing &#x02013; original draft. TA: Data curation, Investigation, Writing &#x02013; original draft. TC: Supervision, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. ST: Conceptualization, Formal analysis, Supervision, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. PB: Investigation, Supervision, Validation, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing.</p></sec>
</body>
<back>
<sec sec-type="funding-information" id="s10">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article.</p>
</sec>
<ack><p>This work was supported by Innovation Research and Development at Kasikorn Business-Technology Group (KBTG).</p>
</ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>CS and TC were employed at Kasikorn Labs, Kasikorn Business-Technology Group. The authors declare that this study received funding from Kasikorn Business-Technology Group. The funder had the following involvement in the study: study design, data collection and analysis. The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s12">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/frai.2024.1361483/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/frai.2024.1361483/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.PDF" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/></sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Agarwal</surname> <given-names>O.</given-names></name> <name><surname>Ge</surname> <given-names>H.</given-names></name> <name><surname>Shakeri</surname> <given-names>S.</given-names></name> <name><surname>Al-Rfou</surname> <given-names>R.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Knowledge graph based synthetic corpus generation for knowledge-enhanced language model pre-training,&#x0201D;</article-title> in <source>Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies</source>, <fpage>3554</fpage>&#x02013;<lpage>3565</lpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ahmad</surname> <given-names>W. U.</given-names></name> <name><surname>Peng</surname> <given-names>N.</given-names></name> <name><surname>Chang</surname> <given-names>K.-W.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;GATE: graph attention transformer encoder for cross-lingual relation and event extraction,&#x0201D;</article-title> in <source>Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 35</source>, <fpage>12462</fpage>&#x02013;<lpage>12470</lpage>.</citation>
</ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Asai</surname> <given-names>A.</given-names></name> <name><surname>Yu</surname> <given-names>X.</given-names></name> <name><surname>Kasai</surname> <given-names>J.</given-names></name> <name><surname>Hajishirzi</surname> <given-names>H.</given-names></name></person-group> (<year>2021</year>). <article-title>One question answering model for many languages with cross-lingual dense passage retrieval</article-title>. <source>Adv. Neural Inform. Process. Syst</source>. <volume>34</volume>, <fpage>7547</fpage>&#x02013;<lpage>7560</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.2107.11976</pub-id></citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bekoulis</surname> <given-names>G.</given-names></name> <name><surname>Deleu</surname> <given-names>J.</given-names></name> <name><surname>Demeester</surname> <given-names>T.</given-names></name> <name><surname>Develder</surname> <given-names>C.</given-names></name></person-group> (<year>2018</year>). <article-title>Joint entity recognition and relation extraction as a multi-head selection problem</article-title>. <source>Expert Syst. Appl</source>. <volume>114</volume>, <fpage>34</fpage>&#x02013;<lpage>45</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1804.07847</pub-id></citation>
</ref>
<ref id="B5">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Bj&#x000F6;rne</surname> <given-names>J.</given-names></name> <name><surname>Salakoski</surname> <given-names>T.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Biomedical event extraction using convolutional neural networks and dependency parsing,&#x0201D;</article-title> in <source>Proceedings of the BioNLP 2018 Workshop</source> (<publisher-loc>Melbourne, VIC</publisher-loc>), <fpage>98</fpage>&#x02013;<lpage>108</lpage>.</citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bronstein</surname> <given-names>O.</given-names></name> <name><surname>Dagan</surname> <given-names>I.</given-names></name> <name><surname>Li</surname> <given-names>Q.</given-names></name> <name><surname>Ji</surname> <given-names>H.</given-names></name> <name><surname>Frank</surname> <given-names>A.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Seed-based event trigger labeling: How far can event descriptions get us?,&#x0201D;</article-title> in <source>Proceedings of the 53rd Annual Meeting of the Association for Computational Linguistics and the 7th International Joint Conference on Natural Language Processing (Vol. 2: Short Papers)</source>, eds. C. Zong and M. Strube (Beijing: Association for Computational Linguistics), <fpage>372</fpage>&#x02013;<lpage>376</lpage>.</citation>
</ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brown</surname> <given-names>T.</given-names></name> <name><surname>Mann</surname> <given-names>B.</given-names></name> <name><surname>Ryder</surname> <given-names>N.</given-names></name> <name><surname>Subbiah</surname> <given-names>M.</given-names></name> <name><surname>Kaplan</surname> <given-names>J. D.</given-names></name> <name><surname>Dhariwal</surname> <given-names>P.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Language models are few-shot learners</article-title>. <source>Adv. Neural Inform. Process. Syst</source>. <volume>33</volume>, <fpage>1877</fpage>&#x02013;<lpage>1901</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.2005.14165</pub-id></citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cassidy</surname> <given-names>T.</given-names></name> <name><surname>McDowell</surname> <given-names>B.</given-names></name> <name><surname>Chambers</surname> <given-names>N.</given-names></name> <name><surname>Bethard</surname> <given-names>S.</given-names></name></person-group> (<year>2014</year>). <article-title>&#x0201C;An annotation framework for dense event ordering,&#x0201D;</article-title> in <source>Proceedings of the 52nd Annual Meeting of the Association for Computational Linguistics (Vol. 2: Short Papers)</source>, eds. K. Toutanova and H. Wu (Baltimore, MD: Association for Computational Linguistics), <fpage>501</fpage>&#x02013;<lpage>506</lpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chambers</surname> <given-names>N.</given-names></name> <name><surname>Cassidy</surname> <given-names>T.</given-names></name> <name><surname>McDowell</surname> <given-names>B.</given-names></name> <name><surname>Bethard</surname> <given-names>S.</given-names></name></person-group> (<year>2014</year>). <article-title>Dense event ordering with a multi-pass architecture</article-title>. <source>Trans. Assoc. Comput. Linguist</source>. <volume>2</volume>, <fpage>273</fpage>&#x02013;<lpage>284</lpage>. <pub-id pub-id-type="doi">10.1162/tacl_a_00182</pub-id></citation>
</ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chau</surname> <given-names>M. T.</given-names></name> <name><surname>Esteves</surname> <given-names>D.</given-names></name> <name><surname>Lehmann</surname> <given-names>J.</given-names></name></person-group> (<year>2019</year>). <article-title>Open-domain event extraction and embedding for natural gas market prediction</article-title>. <source>arXiv preprint arXiv:1912.11334</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1912.11334</pub-id></citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>L.</given-names></name> <name><surname>Ruan</surname> <given-names>W.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Lu</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;SeqVAT: Virtual adversarial training for semi-supervised sequence labeling,&#x0201D;</article-title> in <source>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics</source>, <fpage>8801</fpage>&#x02013;<lpage>8811</lpage>.</citation>
</ref>
<ref id="B12">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Xu</surname> <given-names>L.</given-names></name> <name><surname>Liu</surname> <given-names>K.</given-names></name> <name><surname>Zeng</surname> <given-names>D.</given-names></name> <name><surname>Zhao</surname> <given-names>J.</given-names></name></person-group> (<year>2015a</year>). <article-title>&#x0201C;Event extraction via dynamic multi-pooling convolutional neural networks,&#x0201D;</article-title> in <source>Proceedings of the 53rd Annual Meeting of the Association for Computational Linguistics and the 7th International Joint Conference on Natural Language Processing (Vol. 1: Long Papers)</source> (<publisher-loc>Beijing</publisher-loc>), <fpage>167</fpage>&#x02013;<lpage>176</lpage>.</citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Xu</surname> <given-names>L.</given-names></name> <name><surname>Liu</surname> <given-names>K.</given-names></name> <name><surname>Zeng</surname> <given-names>D.</given-names></name> <name><surname>Zhao</surname> <given-names>J.</given-names></name></person-group> (<year>2015b</year>). <article-title>&#x0201C;Event extraction via dynamic multi-pooling convolutional neural networks,&#x0201D;</article-title> in <source>Proceedings of the 53rd Annual Meeting of the Association for Computational Linguistics and the 7th International Joint Conference on Natural Language Processing (Vol. 1: Long Papers)</source>, eds. C. Zong and M. Strube (Beijing: Association for Computational Linguistics), <fpage>167</fpage>&#x02013;<lpage>176</lpage>.</citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chung</surname> <given-names>H. W.</given-names></name> <name><surname>Hou</surname> <given-names>L.</given-names></name> <name><surname>Longpre</surname> <given-names>S.</given-names></name> <name><surname>Zoph</surname> <given-names>B.</given-names></name> <name><surname>Tay</surname> <given-names>Y.</given-names></name> <name><surname>Fedus</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Scaling instruction-finetuned language models</article-title>. <source>arXiv preprint arXiv:2210.11416</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2210.11416</pub-id></citation>
</ref>
<ref id="B15">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Colin</surname> <given-names>E.</given-names></name> <name><surname>Gardent</surname> <given-names>C.</given-names></name> <name><surname>M&#x00027;rabet</surname> <given-names>Y.</given-names></name> <name><surname>Narayan</surname> <given-names>S.</given-names></name> <name><surname>Perez-Beltrachini</surname> <given-names>L.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;The webnlg challenge: generating text from dbpedia data,&#x0201D;</article-title> in <source>Proceedings of the 9th International Natural Language Generation Conference</source> (<publisher-loc>Edinburgh</publisher-loc>), <fpage>163</fpage>&#x02013;<lpage>167</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Devlin</surname> <given-names>J.</given-names></name> <name><surname>Chang</surname> <given-names>M.-W.</given-names></name> <name><surname>Lee</surname> <given-names>K.</given-names></name> <name><surname>Toutanova</surname> <given-names>K.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;BERT: pre-training of deep bidirectional transformers for language understanding,&#x0201D;</article-title> in <source>Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Vol. 1 (Long and Short Papers)</source>, eds. J. Burstein, C. Doran, and T. Solorio (Minneapolis, MN: Association for Computational Linguistics), <fpage>4171</fpage>&#x02013;<lpage>4186</lpage>.</citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Eberts</surname> <given-names>M.</given-names></name> <name><surname>Ulges</surname> <given-names>A.</given-names></name></person-group> (<year>2019</year>). <article-title>Span-based joint entity and relation extraction with transformer pre-training</article-title>. <source>arXiv preprint arXiv:1909.07755</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1909.07755</pub-id></citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Elhammadi</surname> <given-names>S.</given-names></name> <name><surname>Lakshmanan</surname> <given-names>L. V.</given-names></name> <name><surname>Ng</surname> <given-names>R.</given-names></name> <name><surname>Simpson</surname> <given-names>M.</given-names></name> <name><surname>Huai</surname> <given-names>B.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>&#x0201C;A high precision pipeline for financial knowledge graph construction,&#x0201D;</article-title> in <source>Proceedings of the 28th International Conference on Computational Linguistics</source>, <fpage>967</fpage>&#x02013;<lpage>977</lpage>.</citation>
</ref>
<ref id="B19">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Elsahar</surname> <given-names>H.</given-names></name> <name><surname>Vougiouklis</surname> <given-names>P.</given-names></name> <name><surname>Remaci</surname> <given-names>A.</given-names></name> <name><surname>Gravier</surname> <given-names>C.</given-names></name> <name><surname>Hare</surname> <given-names>J.</given-names></name> <name><surname>Laforest</surname> <given-names>F.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>&#x0201C;T-REx: a large scale alignment of natural language with knowledge base triples,&#x0201D;</article-title> in <source>Proceedings of the Eleventh International Conference on Language Resources and Evaluation (LREC 2018)</source> (<publisher-loc>Miyazaki</publisher-loc>).</citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fader</surname> <given-names>A.</given-names></name> <name><surname>Soderland</surname> <given-names>S.</given-names></name> <name><surname>Etzioni</surname> <given-names>O.</given-names></name></person-group> (<year>2011</year>). <article-title>&#x0201C;Identifying relations for open information extraction,&#x0201D;</article-title> in <source>Proceedings of the 2011 Conference on Empirical Methods in Natural Language Processing</source>, eds. R. Barzilay and M. Johnson (Edinburgh: Association for Computational Linguistics), <fpage>1535</fpage>&#x02013;<lpage>1545</lpage>.</citation>
</ref>
<ref id="B21">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Gupta</surname> <given-names>S.</given-names></name> <name><surname>Manning</surname> <given-names>C. D.</given-names></name></person-group> (<year>2014</year>). <article-title>&#x0201C;Improved pattern learning for bootstrapped entity extraction,&#x0201D;</article-title> in <source>Proceedings of the Eighteenth Conference on Computational Natural Language Learning</source> (<publisher-loc>Ann Arbor, MI</publisher-loc>), <fpage>98</fpage>&#x02013;<lpage>108</lpage>.</citation>
</ref>
<ref id="B22">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Guu</surname> <given-names>K.</given-names></name> <name><surname>Lee</surname> <given-names>K.</given-names></name> <name><surname>Tung</surname> <given-names>Z.</given-names></name> <name><surname>Pasupat</surname> <given-names>P.</given-names></name> <name><surname>Chang</surname> <given-names>M.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Retrieval augmented language model pre-training,&#x0201D;</article-title> in <source>International Conference on Machine Learning</source> (<publisher-loc>PMLR</publisher-loc>), <fpage>3929</fpage>&#x02013;<lpage>3938</lpage>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Han</surname> <given-names>S.</given-names></name> <name><surname>Hao</surname> <given-names>X.</given-names></name> <name><surname>Huang</surname> <given-names>H.</given-names></name></person-group> (<year>2018</year>). <article-title>An event-extraction approach for business analysis from online chinese news</article-title>. <source>Electr. Commerce Res. Appl</source>. <volume>28</volume>, <fpage>244</fpage>&#x02013;<lpage>260</lpage>. <pub-id pub-id-type="doi">10.1016/j.elerap.2018.02.006</pub-id></citation>
</ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hang</surname> <given-names>T.</given-names></name> <name><surname>Feng</surname> <given-names>J.</given-names></name> <name><surname>Wu</surname> <given-names>Y.</given-names></name> <name><surname>Yan</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name></person-group> (<year>2021</year>). <article-title>Joint extraction of entities and overlapping relations using source-target entity labeling</article-title>. <source>Expert Syst. Appl</source>. <volume>177</volume>:<fpage>114853</fpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2021.114853</pub-id></citation>
</ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hsu</surname> <given-names>I.</given-names></name> <name><surname>Huang</surname> <given-names>K. H.</given-names></name> <name><surname>Boschee</surname> <given-names>E.</given-names></name> <name><surname>Miller</surname> <given-names>S.</given-names></name> <name><surname>Natarajan</surname> <given-names>P.</given-names></name> <name><surname>Chang</surname> <given-names>K. W.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>DEGREE: a data-efficient generative event extraction model</article-title>. <source>arXiv preprint arXiv:2108.12724</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2108.12724</pub-id></citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>L.</given-names></name> <name><surname>Ji</surname> <given-names>H.</given-names></name> <name><surname>Cho</surname> <given-names>K.</given-names></name> <name><surname>Dagan</surname> <given-names>I.</given-names></name> <name><surname>Riedel</surname> <given-names>S.</given-names></name> <name><surname>Voss</surname> <given-names>C.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Zero-shot transfer learning for event extraction,&#x0201D;</article-title> in <source>Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Vol. 1: Long Papers)</source>, eds. I. Gurevych and Y. Miyao (Melbourne, VIC: Association for Computational Linguistics), <fpage>2160</fpage>&#x02013;<lpage>2170</lpage>.<pub-id pub-id-type="pmid">37196988</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>L.</given-names></name> <name><surname>Ji</surname> <given-names>H.</given-names></name> <name><surname>Cho</surname> <given-names>K.</given-names></name> <name><surname>Voss</surname> <given-names>C. R.</given-names></name></person-group> (<year>2017</year>). <article-title>Zero-shot transfer learning for event extraction</article-title>. <source>arXiv preprint arXiv:1707.01066</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1707.01066</pub-id><pub-id pub-id-type="pmid">37196988</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ji</surname> <given-names>Z.</given-names></name> <name><surname>Lee</surname> <given-names>N.</given-names></name> <name><surname>Frieske</surname> <given-names>R.</given-names></name> <name><surname>Yu</surname> <given-names>T.</given-names></name> <name><surname>Su</surname> <given-names>D.</given-names></name> <name><surname>Xu</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Survey of hallucination in natural language generation</article-title>. <source>ACM Comput. Surv</source> <volume>55</volume>, <fpage>1</fpage>&#x02013;<lpage>38</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.2202.03629</pub-id><pub-id pub-id-type="pmid">17156503</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kang</surname> <given-names>M.</given-names></name> <name><surname>Baek</surname> <given-names>J.</given-names></name> <name><surname>Hwang</surname> <given-names>S. J.</given-names></name></person-group> (<year>2022</year>). <article-title>KALA: knowledge-augmented language model adaptation</article-title>. <source>arXiv preprint arXiv:2204.10555</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2204.10555</pub-id></citation>
</ref>
<ref id="B30">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Khongcharoen</surname> <given-names>W.</given-names></name> <name><surname>Saetia</surname> <given-names>C.</given-names></name> <name><surname>Chalothorn</surname> <given-names>T.</given-names></name> <name><surname>Buabthong</surname> <given-names>P.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Question answering over knowledge graphs for thai retail banking products,&#x0201D;</article-title> in <source>Proceeding of The 17th International Joint Symposium on Artificial Intelligence and Natural Language Processing (iSAI-NLP 2022)</source> (<publisher-loc>Chiangmai</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation>
</ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kipf</surname> <given-names>T. N.</given-names></name> <name><surname>Welling</surname> <given-names>M.</given-names></name></person-group> (<year>2017</year>). <article-title>Semi-supervised classification with graph convolutional networks</article-title>. <source>arXiv preprint arXiv:1609.02907</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1609.02907</pub-id></citation>
</ref>
<ref id="B32">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Klie</surname> <given-names>J.-C.</given-names></name> <name><surname>Bugert</surname> <given-names>M.</given-names></name> <name><surname>Boullosa</surname> <given-names>B.</given-names></name> <name><surname>Eckart de Castilho</surname> <given-names>R.</given-names></name> <name><surname>Gurevych</surname> <given-names>I.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;The INCEpTION platform: machine-assisted and knowledge-oriented interactive annotation,&#x0201D;</article-title> in <source>Proceedings of the 27th International Conference on Computational Linguistics: System Demonstrations</source> (<publisher-loc>Santa Fe</publisher-loc>), <fpage>5</fpage>&#x02013;<lpage>9</lpage>.</citation>
</ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lai</surname> <given-names>V. D.</given-names></name> <name><surname>Nguyen</surname> <given-names>T.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Extending event detection to new types with learning from keywords,&#x0201D;</article-title> in <source>Proceedings of the 5th Workshop on Noisy User-generated Text (W-NUT 2019)</source>, eds. W. Xu, A. Ritter, T. Baldwin, and A. Rahimi (Hong Kong: Association for Computational Linguistics), <fpage>243</fpage>&#x02013;<lpage>248</lpage>.</citation>
</ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lewis</surname> <given-names>M.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Goyal</surname> <given-names>N.</given-names></name> <name><surname>Ghazvininejad</surname> <given-names>M.</given-names></name> <name><surname>Mohamed</surname> <given-names>A.</given-names></name> <name><surname>Levy</surname> <given-names>O.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>BART: denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension</article-title>. <source>arXiv preprint arXiv:1910.13461</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1910.13461</pub-id></citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>D.</given-names></name> <name><surname>Huang</surname> <given-names>L.</given-names></name> <name><surname>Ji</surname> <given-names>H.</given-names></name> <name><surname>Han</surname> <given-names>J.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Biomedical event extraction based on knowledge-driven tree-LSTM,&#x0201D;</article-title> in <source>Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Vol. 1 (Long and Short Papers)</source>, eds. J. Burstein, C. Doran, and T. Solorio (Minneapolis, MN: Association for Computational Linguistics), <fpage>1421</fpage>&#x02013;<lpage>1430</lpage>.</citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>F.</given-names></name> <name><surname>Huang</surname> <given-names>R.</given-names></name> <name><surname>Xiong</surname> <given-names>D.</given-names></name> <name><surname>Zhang</surname> <given-names>M.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Learning event expressions via bilingual structure projection,&#x0201D;</article-title> in <source>Proceedings of COLING 2016, the 26th International Conference on Computational Linguistics: Technical Papers</source>, eds. Y. Matsumoto and R. Prasad (Osaka: The COLING 2016 Organizing Committee), <fpage>1441</fpage>&#x02013;<lpage>1450</lpage>.</citation>
</ref>
<ref id="B37">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>F.</given-names></name> <name><surname>Peng</surname> <given-names>W.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>Q.</given-names></name> <name><surname>Pan</surname> <given-names>L.</given-names></name> <name><surname>Lyu</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>&#x0201C;Event extraction as multi-turn question answering,&#x0201D;</article-title> in <source>Findings of the Association for Computational Linguistics: EMNLP 2020</source> (<publisher-loc>EMNLP</publisher-loc>), <fpage>829</fpage>&#x02013;<lpage>838</lpage>.<pub-id pub-id-type="pmid">37321443</pub-id></citation></ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Q.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Sheng</surname> <given-names>J.</given-names></name> <name><surname>Cui</surname> <given-names>S.</given-names></name> <name><surname>Wu</surname> <given-names>J.</given-names></name> <name><surname>Hei</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>A survey on deep learning event extraction: approaches and applications</article-title>. <source>IEEE Trans. Neural Netw. Learn. Syst</source>. <volume>2022</volume>, <fpage>1</fpage>&#x02013;<lpage>21</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2022.3213168</pub-id><pub-id pub-id-type="pmid">36269921</pub-id></citation></ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Min</surname> <given-names>L.</given-names></name> <name><surname>Huang</surname> <given-names>X.</given-names></name></person-group> (<year>2021</year>). <article-title>An overview of event extraction and its applications</article-title>. <source>arXiv preprint arXiv:2111.03212</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2111.03212</pub-id></citation>
</ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Q.</given-names></name> <name><surname>Luan</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>K.</given-names></name> <name><surname>Zou</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>B.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name></person-group> (<year>2023</year>). <article-title>Document-level event extraction&#x02014;a survey of methods and applications</article-title>. <source>J. Phys</source>. <volume>2504</volume>:<fpage>e012008</fpage>. <pub-id pub-id-type="doi">10.1088/1742-6596/2504/1/012008</pub-id></citation>
</ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Huang</surname> <given-names>H.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name></person-group> (<year>2019</year>). <article-title>Open domain event extraction using neural latent variable models</article-title>. <source>arXiv preprint arXiv:1906.06947</source>. <pub-id pub-id-type="doi">10.18653/v1/P19-1276</pub-id><pub-id pub-id-type="pmid">36568019</pub-id></citation></ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Luo</surname> <given-names>Z.</given-names></name> <name><surname>Huang</surname> <given-names>H.</given-names></name></person-group> (<year>2018</year>). <article-title>Jointly multiple events extraction via attention-based graph information aggregation</article-title>. <source>arXiv preprint arXiv:1809.09078</source>. <pub-id pub-id-type="doi">10.18653/v1/D18-1156</pub-id><pub-id pub-id-type="pmid">36568019</pub-id></citation></ref>
<ref id="B43">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Lou</surname> <given-names>C.</given-names></name> <name><surname>Gao</surname> <given-names>J.</given-names></name> <name><surname>Yu</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name> <name><surname>Zhao</surname> <given-names>H.</given-names></name> <name><surname>Tu</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>&#x0201C;Translation-based implicit annotation projection for zero-shot cross-lingual event argument extraction,&#x0201D;</article-title> in <source>Proceedings of the 45th International ACM SIGIR Conference on Research and Development in Information Retrieval, SIGIR &#x00027;22</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>2076</fpage>&#x02013;<lpage>2081</lpage>.</citation>
</ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lowphansirikul</surname> <given-names>L.</given-names></name> <name><surname>Polpanumas</surname> <given-names>C.</given-names></name> <name><surname>Jantrakulchai</surname> <given-names>N.</given-names></name> <name><surname>Nutanong</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>Wangchanberta: pretraining transformer-based thai language models</article-title>. <source>arXiv preprint arXiv:2101.09635</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2101.09635</pub-id></citation>
</ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lu</surname> <given-names>W.</given-names></name> <name><surname>Roth</surname> <given-names>D.</given-names></name></person-group> (<year>2012</year>). <article-title>&#x0201C;Automatic event extraction with structured preference modeling,&#x0201D;</article-title> in <source>Proceedings of the 50th Annual Meeting of the Association for Computational Linguistics (Vol. 1: Long Papers)</source>, eds. H. Li, C. Y. Lin, M. Osborne, G. G. Lee, and J. C. Park (Jeju Island: Association for Computational Linguistics), <fpage>835</fpage>&#x02013;<lpage>844</lpage>.<pub-id pub-id-type="pmid">35871475</pub-id></citation></ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lu</surname> <given-names>Y.</given-names></name> <name><surname>Lin</surname> <given-names>H.</given-names></name> <name><surname>Xu</surname> <given-names>J.</given-names></name> <name><surname>Han</surname> <given-names>X.</given-names></name> <name><surname>Tang</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Text2Event: controllable sequence-to-structure generation for end-to-end event extraction</article-title>. <source>arXiv preprint arXiv:2106.09232</source>. <pub-id pub-id-type="doi">10.18653/v1/2021.acl-long.217</pub-id><pub-id pub-id-type="pmid">36568019</pub-id></citation></ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Luan</surname> <given-names>Y.</given-names></name> <name><surname>He</surname> <given-names>L.</given-names></name> <name><surname>Ostendorf</surname> <given-names>M.</given-names></name> <name><surname>Hajishirzi</surname> <given-names>H.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Multi-task identification of entities, relations, and coreference for scientific knowledge graph construction,&#x0201D;</article-title> in <source>Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing</source>, eds. E. Riloff, D. Chiang, J. Hockenmaier, and J. Tsujii (Brussels: Association for Computational Linguistics), <fpage>3219</fpage>&#x02013;<lpage>3232</lpage>.</citation>
</ref>
<ref id="B48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lyu</surname> <given-names>Q.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Sulem</surname> <given-names>E.</given-names></name> <name><surname>Roth</surname> <given-names>D.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Zero-shot event extraction via transfer learning: challenges and insights,&#x0201D;</article-title> in <source>Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Vol. 2: Short Papers)</source>, eds. C. Zong, F. Xia, W. Li, and R. Navigli (Brussels: Association for Computational Linguistics), <fpage>322</fpage>&#x02013;<lpage>332</lpage>.</citation>
</ref>
<ref id="B49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>M&#x00027;hamdi</surname> <given-names>M.</given-names></name> <name><surname>Freedman</surname> <given-names>M.</given-names></name> <name><surname>May</surname> <given-names>J.</given-names></name></person-group> (<year>2019</year>). &#x0201C;Contextualized cross-lingual event trigger extraction with minimal resources,&#x0201D; in <source>Proceedings of the 23rd Conference on Computational Natural Language Learning (CoNLL)</source>, eds. M. Bansal and A. Villavicencio (Hong Kong: Association for Computational Linguistics), <fpage>656</fpage>&#x02013;<lpage>665</lpage>.</citation>
</ref>
<ref id="B50">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mialon</surname> <given-names>G.</given-names></name> <name><surname>Dess&#x000EC;</surname> <given-names>R.</given-names></name> <name><surname>Lomeli</surname> <given-names>M.</given-names></name> <name><surname>Nalmpantis</surname> <given-names>C.</given-names></name> <name><surname>Pasunuru</surname> <given-names>R.</given-names></name> <name><surname>Raileanu</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Augmented language models: a survey</article-title>. <source>arXiv preprint arXiv:2302.07842</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2302.07842</pub-id></citation>
</ref>
<ref id="B51">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Miller</surname> <given-names>G. A.</given-names></name></person-group> (<year>1995</year>). <article-title>Wordnet: a lexical database for english</article-title>. <source>Commun. ACM</source> <volume>38</volume>, <fpage>39</fpage>&#x02013;<lpage>41</lpage>.</citation>
</ref>
<ref id="B52">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Nguyen</surname> <given-names>T. M.</given-names></name> <name><surname>Nguyen</surname> <given-names>T. H.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;One for all: neural joint modeling of entities and events,&#x0201D;</article-title> in <source>Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 33</source> (<publisher-loc>Hawaii</publisher-loc>), <fpage>6851</fpage>&#x02013;<lpage>6858</lpage>.<pub-id pub-id-type="pmid">32667950</pub-id></citation></ref>
<ref id="B53">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Nivre</surname> <given-names>J.</given-names></name> <name><surname>De Marneffe</surname> <given-names>M. C.</given-names></name> <name><surname>Ginter</surname> <given-names>F.</given-names></name> <name><surname>Goldberg</surname> <given-names>Y.</given-names></name> <name><surname>Hajic</surname> <given-names>J.</given-names></name> <name><surname>Manning</surname> <given-names>C. D.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>&#x0201C;Universal dependencies v1: a multilingual treebank collection,&#x0201D;</article-title> in <source>Proceedings of the Tenth International Conference on Language Resources and Evaluation (LREC&#x00027;16)</source> (<publisher-loc>Portoro&#x0017E;</publisher-loc>), <fpage>1659</fpage>&#x02013;<lpage>1666</lpage>.</citation>
</ref>
<ref id="B54">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pan</surname> <given-names>S.</given-names></name> <name><surname>Luo</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Wu</surname> <given-names>X.</given-names></name></person-group> (<year>2023</year>). <article-title>Unifying large language models and knowledge graphs: a roadmap</article-title>. <source>arXiv preprint arXiv:2306.08302</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2306.08302</pub-id></citation>
</ref>
<ref id="B55">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Paolini</surname> <given-names>G.</given-names></name> <name><surname>Athiwaratkun</surname> <given-names>B.</given-names></name> <name><surname>Krone</surname> <given-names>J.</given-names></name> <name><surname>Ma</surname> <given-names>J.</given-names></name> <name><surname>Achille</surname> <given-names>A.</given-names></name> <name><surname>Anubhai</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Structured prediction as translation between augmented natural languages</article-title>. <source>arXiv preprint arXiv:2101.05779</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2101.05779</pub-id></citation>
</ref>
<ref id="B56">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pyysalo</surname> <given-names>S.</given-names></name> <name><surname>Ohta</surname> <given-names>T.</given-names></name> <name><surname>Rak</surname> <given-names>R.</given-names></name> <name><surname>Sullivan</surname> <given-names>D.</given-names></name> <name><surname>Mao</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>Overview of the ID, EPI and REL tasks of bionlp shared task 2011</article-title>. <source>BMC Bioinformat</source>. 13(Suppl.11):S2. <pub-id pub-id-type="doi">10.1186/1471-2105-13-S11-S2</pub-id><pub-id pub-id-type="pmid">22759456</pub-id></citation></ref>
<ref id="B57">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shen</surname> <given-names>S.</given-names></name> <name><surname>Wu</surname> <given-names>T.</given-names></name> <name><surname>Qi</surname> <given-names>G.</given-names></name> <name><surname>Li</surname> <given-names>Y. F.</given-names></name> <name><surname>Haffari</surname> <given-names>G.</given-names></name> <name><surname>Bi</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Adaptive knowledge-enhanced Bayesian meta-learning for few-shot event detection,&#x0201D;</article-title> in <source>Findings of the Association for Computational Linguistics: ACL-IJCNLP 2021</source>, eds. C. Zong, F. Xia, W. Li, and R. Navigli (Brussels: Association for Computational Linguistics), <fpage>2417</fpage>&#x02013;<lpage>2429</lpage>.</citation>
</ref>
<ref id="B58">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Snell</surname> <given-names>J.</given-names></name> <name><surname>Swersky</surname> <given-names>K.</given-names></name> <name><surname>Zemel</surname> <given-names>R. S.</given-names></name></person-group> (<year>2017</year>). <article-title>Prototypical networks for few-shot learning</article-title>. <source>arXiv preprint arXiv:1703.05175</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1703.05175</pub-id></citation>
</ref>
<ref id="B59">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Speer</surname> <given-names>R.</given-names></name> <name><surname>Chin</surname> <given-names>J.</given-names></name> <name><surname>Havasi</surname> <given-names>C.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;ConceptNet 5.5: an open multilingual graph of general knowledge,&#x0201D;</article-title> in <source>Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 31</source> (<publisher-loc>San Francisco, CA</publisher-loc>).</citation>
</ref>
<ref id="B60">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stenetorp</surname> <given-names>P.</given-names></name> <name><surname>Pyysalo</surname> <given-names>S.</given-names></name> <name><surname>Topi&#x00107;</surname> <given-names>G.</given-names></name> <name><surname>Ohta</surname> <given-names>T.</given-names></name> <name><surname>Ananiadou</surname> <given-names>S.</given-names></name> <name><surname>Tsujii</surname> <given-names>J.</given-names></name></person-group> (<year>2012</year>). <article-title>&#x0201C;brat: a web-based tool for NLP-assisted text annotation,&#x0201D;</article-title> in <source>Proceedings of the Demonstrations at the 13th Conference of the European Chapter of the Association for Computational Linguistics</source>, ed. F. Segond (Avignon: Association for Computational Linguistics), <fpage>102</fpage>&#x02013;<lpage>107</lpage>.</citation>
</ref>
<ref id="B61">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Subburathinam</surname> <given-names>A.</given-names></name> <name><surname>Lu</surname> <given-names>D.</given-names></name> <name><surname>Ji</surname> <given-names>H.</given-names></name> <name><surname>May</surname> <given-names>J.</given-names></name> <name><surname>Chang</surname> <given-names>S.-F.</given-names></name> <name><surname>Sil</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>&#x0201C;Cross-lingual structure transfer for relation and event extraction,&#x0201D;</article-title> in <source>Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)</source>, eds. K. Inui, J. Jiang, V. Ng, and X. Wan (Hong Kong: Association for Computational Linguistics), <fpage>313</fpage>&#x02013;<lpage>325</lpage>.</citation>
</ref>
<ref id="B62">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Tan</surname> <given-names>Q.</given-names></name> <name><surname>Liu</surname> <given-names>N.</given-names></name> <name><surname>Zhao</surname> <given-names>X.</given-names></name> <name><surname>Yang</surname> <given-names>H.</given-names></name> <name><surname>Zhou</surname> <given-names>J.</given-names></name> <name><surname>Hu</surname> <given-names>X.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Learning to hash with graph neural networks for recommender systems,&#x0201D;</article-title> in <source>Proceedings of The Web Conference 2020</source> (<publisher-loc>Taipei</publisher-loc>), <fpage>1988</fpage>&#x02013;<lpage>1998</lpage>.</citation>
</ref>
<ref id="B63">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Tjong Kim Sang</surname> <given-names>E. F.</given-names></name> <name><surname>De Meulder</surname> <given-names>F.</given-names></name></person-group> (<year>2003</year>). <article-title>&#x0201C;Introduction to the CoNLL-2003 shared task: language-independent named entity recognition,&#x0201D;</article-title> in <source>Proceedings of the Seventh Conference on Natural Language Learning at HLT-NAACL 2003</source> (<publisher-loc>Edmonton, AB</publisher-loc>), <fpage>142</fpage>&#x02013;<lpage>147</lpage>.</citation>
</ref>
<ref id="B64">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Touvron</surname> <given-names>H.</given-names></name> <name><surname>Lavril</surname> <given-names>T.</given-names></name> <name><surname>Izacard</surname> <given-names>G.</given-names></name> <name><surname>Martinet</surname> <given-names>X.</given-names></name> <name><surname>Lachaux</surname> <given-names>M.-A.</given-names></name> <name><surname>Lacroix</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>LLaMA: open and efficient foundation language models</article-title>. <source>arXiv preprint arXiv:2302.13971</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2302.13971</pub-id></citation>
</ref>
<ref id="B65">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vanegas</surname> <given-names>J. A.</given-names></name> <name><surname>Matos</surname> <given-names>S.</given-names></name> <name><surname>Gonz&#x000E1;lez</surname> <given-names>F.</given-names></name> <name><surname>Oliveira</surname> <given-names>J. L.</given-names></name></person-group> (<year>2015</year>). <article-title>An overview of biomolecular event extraction from scientific documents</article-title>. <source>Computat. Math. Methods Med</source>. <volume>2015</volume>:<fpage>571381</fpage>. <pub-id pub-id-type="doi">10.1155/2015/571381</pub-id><pub-id pub-id-type="pmid">26587051</pub-id></citation></ref>
<ref id="B66">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Veli&#x0010D;kovi&#x00107;</surname> <given-names>P.</given-names></name> <name><surname>Cucurull</surname> <given-names>G.</given-names></name> <name><surname>Casanova</surname> <given-names>A.</given-names></name> <name><surname>Romero</surname> <given-names>A.</given-names></name> <name><surname>Li&#x000F3;</surname> <given-names>P.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name></person-group> (<year>2018</year>). <article-title>Graph attention networks</article-title>. <source>arXiv preprint arXiv:1710.10903</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1710.10903</pub-id></citation>
</ref>
<ref id="B67">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wadden</surname> <given-names>D.</given-names></name> <name><surname>Wennberg</surname> <given-names>U.</given-names></name> <name><surname>Luan</surname> <given-names>Y.</given-names></name> <name><surname>Hajishirzi</surname> <given-names>H.</given-names></name></person-group> (<year>2019</year>). <article-title>Entity, relation, and event extraction with contextualized span representations</article-title>. <source>arXiv preprint arXiv:1909.03546</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1909.03546</pub-id></citation>
</ref>
<ref id="B68">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Walker</surname> <given-names>C.</given-names></name> <name><surname>Consortium</surname> <given-names>L. D.</given-names></name></person-group> (<year>2005</year>). <source>ACE 2005 Multilingual Training Corpus</source>. Linguistic Data Consortium. Available online at: <ext-link ext-link-type="uri" xlink:href="https://catalog.ldc.upenn.edu/LDC2006T06">https://catalog.ldc.upenn.edu/LDC2006T06</ext-link> (accessed February 1, 2023).</citation>
</ref>
<ref id="B69">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>C.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Chen</surname> <given-names>Z.</given-names></name> <name><surname>Hong</surname> <given-names>H.</given-names></name> <name><surname>Tang</surname> <given-names>J.</given-names></name> <name><surname>Song</surname> <given-names>D.</given-names></name></person-group> (<year>2023</year>). <article-title>DeepStruct: pretraining of language models for structure prediction</article-title>. <source>arXiv preprint arXiv:2205.10475</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2205.10475</pub-id></citation>
</ref>
<ref id="B70">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>T.</given-names></name> <name><surname>Li</surname> <given-names>Z.</given-names></name></person-group> (<year>2023</year>). <article-title>Unsupervised numerical information extraction via exploiting syntactic structures</article-title>. <source>Electronics</source> <volume>12</volume>:<fpage>1977</fpage>. <pub-id pub-id-type="doi">10.3390/electronics12091977</pub-id></citation>
</ref>
<ref id="B71">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>S.</given-names></name> <name><surname>Dredze</surname> <given-names>M.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Are all languages created equal in multilingual BERT?,&#x0201D;</article-title> in <source>Proceedings of the 5th Workshop on Representation Learning for NLP</source>, eds. S. Gella, J. Welbl, M. Rei, F. Petroni, P. Lewis, E. Strubell, et al. (Hong Kong: Association for Computational Linguistics), <fpage>120</fpage>&#x02013;<lpage>130</lpage>.</citation>
</ref>
<ref id="B72">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>S.</given-names></name> <name><surname>Irsoy</surname> <given-names>O.</given-names></name> <name><surname>Lu</surname> <given-names>S.</given-names></name> <name><surname>Dabravolski</surname> <given-names>V.</given-names></name> <name><surname>Dredze</surname> <given-names>M.</given-names></name> <name><surname>Gehrmann</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>BloombergGPT: a large language model for finance</article-title>. <source>arXiv preprint arXiv:2303.17564</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2303.17564</pub-id></citation>
</ref>
<ref id="B73">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xiang</surname> <given-names>W.</given-names></name> <name><surname>Wang</surname> <given-names>B.</given-names></name></person-group> (<year>2019</year>). <article-title>A survey of event extraction from text</article-title>. <source>IEEE Access</source> <volume>7</volume>, <fpage>173111</fpage>&#x02013;<lpage>173137</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2019.2956831</pub-id></citation>
</ref>
<ref id="B74">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xue</surname> <given-names>L.</given-names></name> <name><surname>Constant</surname> <given-names>N.</given-names></name> <name><surname>Roberts</surname> <given-names>A.</given-names></name> <name><surname>Kale</surname> <given-names>M.</given-names></name> <name><surname>Al-Rfou</surname> <given-names>R.</given-names></name> <name><surname>Siddhant</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>&#x0201C;mT5: a massively multilingual pre-trained text-to-text transformer,&#x0201D;</article-title> in <source>Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies</source>, eds. K. Toutanova, A. Rumshisky, L. Zettlemoyer, D. Hakkani-Tur, I. Beltagy, S. Bethard, et al. (Hong Kong: Association for Computational Linguistics), <fpage>483</fpage>&#x02013;<lpage>498</lpage>.</citation>
</ref>
<ref id="B75">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>H.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>K.</given-names></name> <name><surname>Xiao</surname> <given-names>Y.</given-names></name> <name><surname>Zhao</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;DCFEE: a document-level Chinese financial event extraction system based on automatically labeled training data&#x0201D;</article-title> in <source>Proceedings of ACL 2018, System Demonstrations</source> (<publisher-loc>Melbourne, VIC</publisher-loc>), <fpage>50</fpage>&#x02013;<lpage>55</lpage>.</citation>
</ref>
<ref id="B76">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>S.</given-names></name> <name><surname>Feng</surname> <given-names>D.</given-names></name> <name><surname>Qiao</surname> <given-names>L.</given-names></name> <name><surname>Kan</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>D.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Exploring pre-trained language models for event extraction and generation,&#x0201D;</article-title> in <source>Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics</source>, eds. A. Korhonen, D. Traum, and L. M&#x000E0;rquez (Florence: Association for Computational Linguistics), <fpage>5284</fpage>&#x02013;<lpage>5294</lpage>.<pub-id pub-id-type="pmid">32591802</pub-id></citation></ref>
<ref id="B77">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yao</surname> <given-names>L.</given-names></name> <name><surname>Mao</surname> <given-names>C.</given-names></name> <name><surname>Luo</surname> <given-names>Y.</given-names></name></person-group> (<year>2019</year>). <article-title>KG-BERT: BERT for knowledge graph completion</article-title>. <source>arXiv preprint arXiv:1909.03193</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1909.03193</pub-id></citation>
</ref>
<ref id="B78">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Roth</surname> <given-names>D.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Zero-shot label-aware event trigger and argument classification,&#x0201D;</article-title> in <source>Findings of the Association for Computational Linguistics: ACL-IJCNLP 2021</source>, eds. C. Zong, F. Xia, W. Li, and R. Navigli (Florence: Association for Computational Linguistics), <fpage>1331</fpage>&#x02013;<lpage>1340</lpage>.</citation>
</ref>
<ref id="B79">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Qin</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>M.</given-names></name> <name><surname>Ji</surname> <given-names>D.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Extracting entities and events as a single task using a transition-based neural model,&#x0201D;</article-title> in <source>Proceedings of the Twenty-Eighth International Joint Conference on Artificial Intelligence, IJCAI-19</source> (<publisher-loc>Macao</publisher-loc>), <fpage>5422</fpage>&#x02013;<lpage>5428</lpage>.<pub-id pub-id-type="pmid">32667950</pub-id></citation></ref>
</ref-list>
</back>
</article>