<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Electron.</journal-id>
<journal-title>Frontiers in Electronics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Electron.</abbrev-journal-title>
<issn pub-type="epub">2673-5857</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">847069</article-id>
<article-id pub-id-type="doi">10.3389/felec.2022.847069</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Electronics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Hardware-Software Co-Design of an In-Memory Transformer Network Accelerator</article-title>
<alt-title alt-title-type="left-running-head">Laguna et al.</alt-title>
<alt-title alt-title-type="right-running-head">In-Memory Transformer Network Accelerator</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Laguna</surname>
<given-names>Ann Franchesca</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1452597/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Sharifi</surname>
<given-names>Mohammed Mehdi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1733537/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Kazemi</surname>
<given-names>Arman</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1618991/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Yin</surname>
<given-names>Xunzhao</given-names>
</name>
<xref ref-type="corresp" rid="c001">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1162782/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Niemier</surname>
<given-names>Michael</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/709200/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hu</surname>
<given-names>X. Sharon</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1733577/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Computer Science and Engineering</institution>, <institution>University of Notre Dame</institution>, <addr-line>Notre Dame</addr-line>, <addr-line>IN</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>College of Information Science and Electronic Engineering</institution>, <institution>Zhejiang University</institution>, <addr-line>Hangzhou</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1351109/overview">Ram Krishnamurthy</ext-link>, Intel, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1336904/overview">Jae-Sun Seo</ext-link>, Arizona State University, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/474514/overview">Priyadarshini Panda</ext-link>, Yale University, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Ann Franchesca Laguna, <email>alaguna@nd.edu</email>; Xunzhao Yin, <email>xzyin1@zju.edu.cn</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Integrated Circuits and VLSI, a section of the journal Frontiers in Electronics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>11</day>
<month>04</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>3</volume>
<elocation-id>847069</elocation-id>
<history>
<date date-type="received">
<day>01</day>
<month>01</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>14</day>
<month>03</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Laguna, Sharifi, Kazemi, Yin, Niemier and Hu.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Laguna, Sharifi, Kazemi, Yin, Niemier and Hu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Transformer networks have outperformed recurrent and convolutional neural networks in terms of accuracy in various sequential tasks. However, memory and compute bottlenecks prevent transformer networks from scaling to long sequences due to their high execution time and energy consumption. Different neural attention mechanisms have been proposed to lower computational load but still suffer from the memory bandwidth bottleneck. In-memory processing can help alleviate memory bottlenecks by reducing the transfer overhead between the memory and compute units, thus allowing transformer networks to scale to longer sequences. We propose an in-memory transformer network accelerator (iMTransformer) that uses a combination of crossbars and content-addressable memories to accelerate transformer networks. We accelerate transformer networks by (1) computing in-memory, thus minimizing the memory transfer overhead, (2) caching reusable parameters to reduce the number of operations, and (3) exploiting the available parallelism in the attention mechanism computation. To reduce energy consumption, the following techniques are introduced: (1) a configurable attention selector is used to choose different sparse attention patterns, (2) a content-addressable memory aided locality sensitive hashing helps to filter the number of sequence elements by their importance, and (3) FeFET-based crossbars are used to store projection weights while CMOS-based crossbars are used as an attentional cache to store attention scores for later reuse. Using a CMOS-FeFET hybrid iMTransformer introduced a significant energy improvement compared to the CMOS-only iMTransformer. The CMOS-FeFET hybrid iMTransformer achieved an 8.96&#xd7; delay improvement and 12.57&#xd7; energy improvement for the Vanilla transformers compared to the GPU baseline at a sequence length of 512. Implementing BERT using CMOS-FeFET hybrid iMTransformer achieves 13.71&#xd7; delay improvement and 8.95&#xd7; delay improvement compared to the GPU baseline at sequence length of 512. The hybrid iMTransformer also achieves a throughput of 2.23 K samples/sec and 124.8 samples/s/W using the MLPerf benchmark using BERT-large and SQuAD 1.1 dataset, an 11&#xd7; speedup and 7.92&#xd7; energy improvement compared to the GPU baseline.</p>
</abstract>
<kwd-group>
<kwd>transformer network</kwd>
<kwd>processing-in-memory</kwd>
<kwd>accelerator</kwd>
<kwd>sparsity</kwd>
<kwd>crossbars</kwd>
<kwd>CAMs</kwd>
<kwd>FeFET</kwd>
<kwd>CMOS</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Transformer networks (or simply transformers) have continually risen in popularity because of their capability to outperform recurrent neural networks (RNNs) and convolutional neural networks (CNNs), particularly for sequence-based tasks. Different transformer network variants, such as BERT (<xref ref-type="bibr" rid="B12">Devlin et al., 2018</xref>), ALBERT (<xref ref-type="bibr" rid="B32">Lan et al., 2019</xref>), Megatron (<xref ref-type="bibr" rid="B60">Shoeybi et al., 2019</xref>), GPT3 (<xref ref-type="bibr" rid="B4">Brown et al., 2020</xref>), and XLNet (<xref ref-type="bibr" rid="B68">Yang et al., 2019</xref>) currently hold the best performance in various natural language processing (NLP) applications such as machine translation, named entity recognition, and question answering. Transformer networks are also applicable to other sequential tasks such as audio (<xref ref-type="bibr" rid="B3">Boes and Van hamme, 2019</xref>; <xref ref-type="bibr" rid="B66">Yang S.-W. et al., 2020</xref>; <xref ref-type="bibr" rid="B65">Wei et al., 2020</xref>) and video (Boes and Van hamme, 2019; <xref ref-type="bibr" rid="B65">Wei et al., 2020</xref>; <xref ref-type="bibr" rid="B38">Li et al., 2020b</xref>) applications. Transformer networks have also recently been used in computer vision applications (<xref ref-type="bibr" rid="B13">Dosovitskiy et al., 2020</xref>; <xref ref-type="bibr" rid="B40">Liu et al., 2021</xref>). The transformer&#x2019;s superior performance is attributed to the scaled dot-product attention (SDPA) mechanisms that determine the correlation between sequence elements. The attention mechanism in transformer networks can achieve <italic>O</italic> (1) complexity when completely parallelized and can better model long-range dependencies, making them superior to CNNs and RNNs.</p>
<p>Because of the transformer network&#x2019;s capability to learn long-range dependencies, the transformer network can better analyze longer sequence lengths compared to CNNs and RNNs. This leads to the increase in sequence lengths in NLP datasets (<xref ref-type="bibr" rid="B47">Rae et al., 2019</xref>; <xref ref-type="bibr" rid="B59">Sharir et al., 2020</xref>). For example, in language modeling, the Penn Treebank (<xref ref-type="bibr" rid="B41">Marcus et al., 1993</xref>) and WikiText-103 (<xref ref-type="bibr" rid="B42">Merity et al., 2016</xref>) datasets, which are obtained from news and Wikipedia articles, have an average sequence length of 355 and 3.6 K words, respectively. On the other hand, PG-19 (<xref ref-type="bibr" rid="B47">Rae et al., 2019</xref>), a newer dataset for language modeling which is obtained from Project Gutenberg books, has an average sequence length of 69 K words. The use of transformer networks in image and video applications can also contribute to the sequence length explosion. As transformer networks improve and are used in more complex applications, the number of parameters also continues to increase (<xref ref-type="bibr" rid="B59">Sharir et al., 2020</xref>). The vanilla (original) transformer (<xref ref-type="bibr" rid="B63">Vaswani et al., 2017</xref>) began with millions of parameters. Later, transformer network models, e.g., Megatron (<xref ref-type="bibr" rid="B60">Shoeybi et al., 2019</xref>) and GPT-3 (<xref ref-type="bibr" rid="B4">Brown et al., 2020</xref>), contain billions of parameters. Recently, <italic>switch transformers</italic> (<xref ref-type="bibr" rid="B14">Fedus et al., 2021</xref>) used trillions of parameters to account for long-range dependencies in the language model of the PG-19 dataset.</p>
<p>Transformer networks are commonly implemented using general-purpose graphical processing units (GPUs) to exploit the parallelism inherent in the attention mechanism. However, the complexity of implementing the attention mechanism in the GPU is limited to <italic>O</italic> (<italic>dn</italic>
<sup>2</sup>/<italic>c</italic>), where <italic>n</italic> is the sequence length, <italic>d</italic> is the number of feature embedding dimensions, and <italic>c</italic> is the number of parallel cores. Increasing the sequence length and the number of parameters greatly increases the computation latency, memory bandwidth, and energy requirements of transformer networks (<xref ref-type="bibr" rid="B63">Vaswani et al., 2017</xref>) because of the quadratic time and space complexity with respect to the sequence length. Transformer networks with linear time complexity have been proposed (<xref ref-type="bibr" rid="B1">Beltagy et al., 2020</xref>; <xref ref-type="bibr" rid="B75">Zaheer et al., 2020</xref>), but incur the cost of additional space complexity, causing increased memory demand. Moreover, large transformer networks are severely limited by the memory bandwidth. For example, Megatron (<xref ref-type="bibr" rid="B60">Shoeybi et al., 2019</xref>), one of the largest transformer networks to date, only achieves 30% of the theoretical peak FLOPS of a GPU because of the memory bandwidth bottleneck.</p>
<p>Different techniques have been proposed to alleviate problems associated with the explosion in memory and time complexity. These techniques include model parallelism using multi-GPUs (<xref ref-type="bibr" rid="B60">Shoeybi et al., 2019</xref>), caching attention weights (<xref ref-type="bibr" rid="B1">Beltagy et al., 2020</xref>), cross-layer parameter sharing (<xref ref-type="bibr" rid="B32">Lan et al., 2019</xref>), model compression (<xref ref-type="bibr" rid="B74">Zafrir et al., 2019</xref>; <xref ref-type="bibr" rid="B39">Li et al., 2020c</xref>), and sparsification (<xref ref-type="bibr" rid="B8">Child et al., 2019</xref>; <xref ref-type="bibr" rid="B27">Kitaev et al., 2020</xref>; <xref ref-type="bibr" rid="B14">Fedus et al., 2021</xref>). Model parallelism (<xref ref-type="bibr" rid="B60">Shoeybi et al., 2019</xref>) further exacerbates the memory bandwidth bottleneck because of sparse random memory accesses and communication between different GPUs. Transformer network model compression is implemented <italic>via</italic> cross-layer parameter sharing (<xref ref-type="bibr" rid="B32">Lan et al., 2019</xref>), quantization (<xref ref-type="bibr" rid="B74">Zafrir et al., 2019</xref>), and pruning (<xref ref-type="bibr" rid="B39">Li et al., 2020c</xref>). Transformer network sparsity can either be (temporal) locality-based (<xref ref-type="bibr" rid="B8">Child et al., 2019</xref>; <xref ref-type="bibr" rid="B1">Beltagy et al., 2020</xref>) or content-based (<xref ref-type="bibr" rid="B27">Kitaev et al., 2020</xref>; <xref ref-type="bibr" rid="B54">Roy et al., 2021</xref>). An example of content-based sparsity is using locality-sensitive hashing (LSH), an approximate nearest neighbor search algorithm that hashes nearby points to the same hash signature. However, these techniques do not solve the memory bandwidth bottleneck problem.</p>
<p>Processing-in-memory (PIM) (<xref ref-type="bibr" rid="B43">Mutlu et al., 2020</xref>; <xref ref-type="bibr" rid="B56">Sebastian et al., 2020</xref>) has been proposed to solve the memory bandwidth bottleneck by eliminating the communication overhead between the compute unit and the memory. PIM has been used in various applications such as few-shot learning (<xref ref-type="bibr" rid="B44">Ni et al., 2019</xref>; <xref ref-type="bibr" rid="B49">Ranjan et al., 2019</xref>; <xref ref-type="bibr" rid="B5">Challapalle et al., 2020</xref>; <xref ref-type="bibr" rid="B51">Reis et al., 2021</xref>), DNA assembly (<xref ref-type="bibr" rid="B22">Kaplan et al., 2018</xref>; <xref ref-type="bibr" rid="B16">Huangfu et al., 2018</xref>; <xref ref-type="bibr" rid="B29">Laguna et al., 2020</xref>), and security (<xref ref-type="bibr" rid="B53">Reis et al., 2020b</xref>). In particular, PIM-based attention mechanisms have been proposed using content-addressable memories (CAMs) (<xref ref-type="bibr" rid="B31">Laguna et al., 2019a</xref>,<xref ref-type="bibr" rid="B30">b</xref>), crossbar arrays (<xref ref-type="bibr" rid="B49">Ranjan et al., 2019</xref>; <xref ref-type="bibr" rid="B5">Challapalle et al., 2020</xref>), and general-purpose computing-in-memory arrays (GP-CiM) (<xref ref-type="bibr" rid="B50">Reis et al., 2020a</xref>). CAMs can perform fast parallel searches in a single cycle, while crossbar arrays can perform matrix-vector multiplications in a single cycle. GP-CiM can perform bitwise and arithmetic operations in memory. However, PIM-based attention mechanisms have primarily focused on recurrent and memory augmented neural networks (MANNs). Transformer networks that use SDPA have more parallelization opportunities than recurrent and MANN-based attention mechanisms. The SDPA used in transformer networks can be efficiently implemented using crossbar arrays to perform matrix-vector multiplications. CAM arrays can be used to implement content-based sparse attention <italic>via</italic> LSH.</p>
<p>PIM architectures are either based on complementary metal-oxide-semiconductor (CMOS) memories or emerging technologies (<xref ref-type="bibr" rid="B17">Jeloka et al., 2016</xref>; <xref ref-type="bibr" rid="B21">Kang et al., 2017</xref>; <xref ref-type="bibr" rid="B76">Zhang et al., 2017</xref>; <xref ref-type="bibr" rid="B52">Reis et al., 2018</xref>; <xref ref-type="bibr" rid="B49">Ranjan et al., 2019</xref>; <xref ref-type="bibr" rid="B71">Yin et al., 2019</xref>). While improvements in CMOS technology due to transistor scaling have continuously reduced the cost of on-chip and off-chip memories, PIM devices implemented in CMOS technology have low density and high leakage power and require periodic data refreshing. This makes it difficult to apply CMOS solutions to large, data-centric workloads. Alternatively, non-volatile memories (NVM) based on emerging technologies such as ferroelectric field-effect transistors (FeFET), resistive memories (ReRAM), and phase change memories (PCM) have high density, consume low power, and are non-volatile. However, NVMs require higher write times and energy than CMOS technology, making them less ideal for high-write scenarios. FeFETs are CMOS compatible and have been co-integrated in CMOS platforms by GlobalFoundries (<xref ref-type="bibr" rid="B2">Beyer et al., 2020</xref>).</p>
<p>In this paper, we present iMTransformer, an in-memory computing-based accelerator for transformer network inference. iMTransformer employs a combination of crossbars and CAMs. We also use algorithm-based techniques to improve the latency and energy consumption of iMTransformer. iMTransformer reduces the execution time by (1) mitigating the memory-bandwidth bottleneck with processing-in-memory-based hardware, (2) reducing the computational requirements <italic>via</italic> data reuse using attention caches, and (3) maximizing the parallelism that can be achieved with different types of attention mechanisms. iMTransformer further improves energy efficiency by (1) employing an attention selector that can implement masked attention and locality-based sparse attention, (2) using CAMs to implement content-based sparsity through LSH, and (3) using non-volatile FeFET-based crossbars for high-read sublayers and write-efficient CMOS-based crossbars for high-write sublayers.</p>
<p>The standard CMOS implementation of iMTransformer achieves a delay improvement of 7.7&#xd7; and an energy improvement of 7.81&#xd7; compared to the GPU baseline for a sequence with length 512 by using PIM. After including model parallelization, sparsity, and using CMOS-FeFET hybrid implementation, iMTransformer achieves 8.96&#xd7; delay improvement and 12.58&#xd7; energy improvement compared to the GPU baseline. Furthermore, implementing BERT achieves a delay improvement of 4.68&#xd7; for the standard implementation and 13.71&#xd7; after including model parallelization, sparsity, and CMOS-FeFET hybrid iMTransformer implementation. The BERT energy improvement is 4.78&#xd7; for the standard implementation and 8.95&#xd7; for the implementation with model parallelization, sparsity, and using CMOS-FeFET hybrid iMTransformer implementation. The hybrid iMTransformer can process 2.23 K samples/s and 125 samples/s/W of the SQuAD 1.1 dataset using BERT-large and achieves an end-to-end improvement of 11&#xd7; for the delay and 7.92&#xd7; for the energy compared to the GPU baseline.</p>
</sec>
<sec id="s2">
<title>2 Background</title>
<p>Transformer networks (discussed in <xref ref-type="sec" rid="s2-1">Section 2.1</xref>) currently hold the state of the art accuracy in NLP, computer vision, and various fields and have the capacity to model long-range dependencies (<xref ref-type="bibr" rid="B62">Tay et al., 2020b</xref>). That said, the impressive results achieved by transformer networks come with high computational and memory costs as the sequence length increases. Algorithm-based solutions that aim to reduce the space and the computational complexity of transformer networks are presented in <xref ref-type="sec" rid="s2-2">Section 2.2</xref>. These algorithm-based solutions, however, do not solve the memory-bandwidth bottleneck problem. We propose using a PIM-based solution to remove the need for massive data transfers. The PIM-based computing kernels are presented in <xref ref-type="sec" rid="s2-3">Section 2.3</xref>.</p>
<sec id="s2-1">
<title>2.1 Transformer Networks</title>
<p>Transformer networks have outperformed CNNs and RNNs in various tasks because of their ability to model long-range dependencies through attention mechanisms. Because of this, the transformer networks have been used in various applications such as machine translation, text generation, and language modeling. These different applications have led to different transformer designs. <xref ref-type="sec" rid="s2-1-1">Section 2.1.1</xref> discusses a key component of transformer networks: the attention mechanism, particularly scaled-dot product attention (SDPA) and multi-head attention (MHA). <xref ref-type="sec" rid="s2-1-2">Section 2.1.2</xref> then discusses the different types of transformer networks, while <xref ref-type="sec" rid="s2-1-3">Section 2.1.3</xref> considers the execution time distribution of transformer networks.</p>
<sec id="s2-1-1">
<title>2.1.1 Attention</title>
<p>Attention, a crucial component of human intelligence, allows humans to determine the most relevant parts of a sequence (i.e., text) or object and pay less attention to less relevant parts. Neural attention mechanisms work similarly to the human attention mechanism, where more important regions are given more attentional weights than less important ones. Transformer networks rely on SDPA (<xref ref-type="fig" rid="F1">Figure 1A</xref>) to determine the relationship between sequence elements. SDPA uses a key-value-based retrieval where each key corresponds to a value. The query vector <italic>q</italic> is compared to a set of <italic>n</italic> key vectors <bold>K</bold> &#x3d; {<italic>k</italic>
<sub>1</sub>, <italic>k</italic>
<sub>2</sub>, <italic>k</italic>
<sub>3</sub>, &#x2026; , <italic>k</italic>
<sub>
<italic>n</italic>
</sub>} to retrieve similar values in the set of <italic>n</italic> value vectors <bold>V</bold> &#x3d; {<italic>v</italic>
<sub>1</sub>, <italic>v</italic>
<sub>2</sub>, <italic>v</italic>
<sub>3</sub>, &#x2026; , <italic>v</italic>
<sub>
<italic>n</italic>
</sub>}. The SDPA is then calculated as a linear combination of value vectors weighted by the scaled probability distribution of the similarity between <italic>q</italic> and each <italic>k</italic>
<sub>
<italic>j</italic>
</sub> in <bold>K</bold>. To keep the variance equal to one, the dot-product attention is scaled by the number of dimensions of the key vectors <italic>d</italic>
<sub>
<italic>k</italic>
</sub>.<disp-formula id="e1">
<mml:math id="m1">
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold">K</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold">V</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold">K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
<mml:mi mathvariant="bold">V</mml:mi>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Transformers use scaled dot product attention <bold>(A)</bold> and multi-head attention <bold>(B)</bold> to model long-range dependencies and use an encoder-decoder architecture (<xref ref-type="bibr" rid="B63">Vaswani et al., 2017</xref>). The encoder <bold>(C)</bold> and decoder <bold>(D)</bold> layers are composed of multihead attention, feedforward and normalization sublayers. The execution time distribution <bold>(E)</bold> shows the transformer networks becomes more dominated by the multi-head attention as sequence length increases. The execution time <bold>(F)</bold> of transformers increases as the sequence length increases.</p>
</caption>
<graphic xlink:href="felec-03-847069-g001.tif"/>
</fig>
<p>Transformer networks also introduced the concept of MHA (<xref ref-type="fig" rid="F1">Figure 1B</xref>) that allows the network to look at the input in different subspaces. MHA allows transformer networks to analyze the relationships among sequence elements in a highly parallelizable manner. In MHA, the feature embedding is projected into different subspaces (one subspace per head) where the sequence elements can be attended in parallel. Each head (or subspace) can reveal different information regarding the input. The <italic>i</italic>-th head projects the query vector <italic>q</italic>&#x2032;, the set of key vectors <bold>K</bold>&#x2032;, and the set of value vectors <bold>V</bold>&#x2032; by using projection matrices <inline-formula id="inf1">
<mml:math id="m2">
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">Q</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">K</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">V</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> before calculating the SDPA. The output of the SDPA for each attention head is then concatenated and projected by a linear layer.</p>
</sec>
<sec id="s2-1-2">
<title>2.1.2 Encoder and Decoder Layers of Transformer Networks</title>
<p>The (original) vanilla transformer (<xref ref-type="bibr" rid="B63">Vaswani et al., 2017</xref>) is a neural network model that uses an encoder-decoder architecture (<xref ref-type="fig" rid="F1">Figures 1C,D</xref>). The encoder layer (<xref ref-type="fig" rid="F1">Figure 1C</xref>) accepts a variable-length input and transforms it to a fixed-length feature vector. The decoder layer (<xref ref-type="fig" rid="F1">Figure 1D</xref>) accepts this fixed-length feature vector and then transforms it into a variable-length feature vector. This allows the network to accept inputs of varying lengths. Sequence-to-sequence models (<xref ref-type="bibr" rid="B63">Vaswani et al., 2017</xref>; <xref ref-type="bibr" rid="B19">Junczys-Dowmunt et al., 2018</xref>; <xref ref-type="bibr" rid="B34">Lewis et al., 2019</xref>; <xref ref-type="bibr" rid="B48">Raffel et al., 2019</xref>) follow the encoder-decoder structure of the vanilla transformer (<xref ref-type="bibr" rid="B63">Vaswani et al., 2017</xref>) and are commonly used in machine translation tasks, question answering, and summarization. Other applications, however, require different topologies. Some applications (embedding to sequence), such as text generation, only require decoder layers. Others, such as language modeling and sentence classification (sequence to embedding), only require encoder layers. Transformer networks hence can be categorized into (1) sequence-to-sequence models, (2) decoder-only (or auto-regressive) models, and (3) encoder-only (auto-encoding) models.</p>
<p>Decoder-only transformer networks are naturally used in text and image generation. Decoder-only (auto-regressive) transformer models, such as GPT-2 (<xref ref-type="bibr" rid="B46">Radford et al., 2019</xref>) and Transformer-XL (<xref ref-type="bibr" rid="B11">Dai et al., 2019</xref>), use masked MHA where the attention scores of the current input are only based on attention scores of past inputs and not on future inputs. Because of their auto-regressive nature, caching the attention scores (keys and values) can reduce the computational requirements of each time step at the cost of higher storage demands (<xref ref-type="bibr" rid="B11">Dai et al., 2019</xref>).</p>
<p>Encoder-only transformer networks are usually used for language modeling and sentence/token classification. Encoder-only (auto-encoding) transformer models, such as BERT (<xref ref-type="bibr" rid="B12">Devlin et al., 2018</xref>) and ALBERT (<xref ref-type="bibr" rid="B32">Lan et al., 2019</xref>), do not use masking, and each input is influenced by past and future inputs (bidirectional). Due to their bidirectional nature, auto-encoding transformer models can be greatly parallelized <italic>via</italic> data and model parallelization (<xref ref-type="bibr" rid="B60">Shoeybi et al., 2019</xref>).</p>
</sec>
<sec id="s2-1-3">
<title>2.1.3 Transformer Network Execution Time</title>
<p>To investigate the execution time needed by different functions in the encoder and decoder layers of transformer networks, we profiled the Vanilla Transformer running on a Titan X GPU. <xref ref-type="fig" rid="F1">Figures 1E,F</xref> shows the resulting execution time distribution of the transformer network. The encoder and decoder layers of transformer networks are primarily composed of MHA, feedforward, and normalization sublayers, as shown in <xref ref-type="fig" rid="F1">Figures 1C,D</xref>. As the sequence length <italic>n</italic> increases, the execution time of the transformer network increases quadratically <italic>O</italic> (<italic>n</italic>
<sup>2</sup>) due to the MHA. The feedforward layers only increase linearly <italic>O</italic>(<italic>n</italic>). Because of this, the MHA dominates the execution time of the transformer network as the sequence length increases, as shown in <xref ref-type="fig" rid="F1">Figure 1E</xref>.</p>
<p>In this work, we focus on accelerating MHA. In particular, different types of transformer networks have different MHA properties, which can be exploited when designing transformer network accelerators. For example, decoder-only transformer models require masked MHA. By not executing the masked operations, the number of computations can be reduced. The encoder-only transformer models use bidirectional MHA. By exploiting model parallelism for bidirectional MHA, transformer networks can be accelerated. We utilize these transformer network properties in designing our transformer network accelerator.</p>
</sec>
</sec>
<sec id="s2-2">
<title>2.2 Algorithm-Based Transformer Network Acceleration</title>
<p>Transformer networks have high computational and space complexity because of the employed attention mechanism. Transformer network GPU implementations are bounded by the memory, particularly with longer sequences, because of the <italic>O</italic> (<italic>dn</italic> &#x2b; <italic>dn</italic>
<sup>2</sup>) spatial complexity, where <italic>d</italic> represents the feature embedding dimension, and <italic>n</italic> is the sequence length. A transformer can also be computation-limited because of the<italic>O</italic> (<italic>dn</italic>
<sup>2</sup>) serialized time complexity of the MHA. The <italic>O</italic> (<italic>dn</italic>
<sup>2</sup>) complexity comes from each time step (sequence element) attending to every other time step (sequence element). However, an <italic>O</italic> (1) time complexity can be achieved with adequate parallelism. The following sections review four types of algorithm-based acceleration: quantization (<xref ref-type="sec" rid="s2-2-1">Section 2.2.1</xref>), attention caching (<xref ref-type="sec" rid="s2-2-2">Section 2.2.2</xref>), model parallelism (<xref ref-type="sec" rid="s2-2-3">Section 2.2.3</xref>) and sparse attention (<xref ref-type="sec" rid="s2-2-4">Section 2.2.4</xref>).</p>
<sec id="s2-2-1">
<title>2.2.1 Quantization</title>
<p>Reducing the representation precision by quantization can alleviate the memory demand and reduce the time complexity of transformer networks by reducing the amount of data transfer required between the compute and memory units. FullyQT and Q8BERT studied transformer quantization. FullyQT (<xref ref-type="bibr" rid="B45">Prato et al., 2019</xref>) used <italic>k</italic>-bit uniform quantization. Transformer networks quantized in 8-bits performed better in 21 out of 35 experiments made in FullyQT, and there was minimal degradation on the other experiments. Q8BERT (<xref ref-type="bibr" rid="B74">Zafrir et al., 2019</xref>) used quantization-aware training and has shown that it performs better than using dynamic quantization. Both FullyQT and Q8BERT have found that 8-bit quantization is found to have comparable accuracy to transformer networks at full precision.</p>
<p>Non-uniform quantization has also been proposed for transformer networks (<xref ref-type="bibr" rid="B10">Chung et al., 2020</xref>) and has shown better compression without sacrificing accuracy. These proposed non-uniform quantization techniques decouple feature vectors into a set of weights and binary vectors. These binary codes are also more hardware-friendly than other quantization methods.</p>
</sec>
<sec id="s2-2-2">
<title>2.2.2 Attention Caching</title>
<p>The decoder layer of the transformer is auto-regressive (i.e., the current output is the next input). The attention keys and values from previous time steps affect the current time step. These keys and values have been computed in previous time steps and can be reused instead of recomputing from scratch as implemented in the vanilla transformer (<xref ref-type="bibr" rid="B63">Vaswani et al., 2017</xref>). Transformer-XL (<xref ref-type="bibr" rid="B11">Dai et al., 2019</xref>) implemented attention caching and achieved up to 1800&#xd7; improvement during inference at a sequence length of 3.8 K. <italic>Attention caching</italic> improves the computational complexity of the transformer during inference at the cost of higher space complexity. Furthermore, attention caching allows the modeling of a longer range of dependencies by using in conjunction with sparse attention (<xref ref-type="bibr" rid="B8">Child et al., 2019</xref>; <xref ref-type="bibr" rid="B11">Dai et al., 2019</xref>). Sparse attention is explained in detail in <xref ref-type="sec" rid="s2-2-4">Section 2.2.4</xref>. We use attention caching in designing our transformer network accelerator.</p>
</sec>
<sec id="s2-2-3">
<title>2.2.3 Model Parallelism</title>
<p>
<italic>Model parallelism</italic> aims to distribute a neural network model into multiple compute units to reduce time complexity and improve compute unit utilization. Megatron (<xref ref-type="bibr" rid="B60">Shoeybi et al., 2019</xref>) used model parallelism to split different attention heads into multiple GPUs. By implementing model parallelism. Megatron improved the GPU utilization from 30% theoretical peak FLOPS for a single GPU to 52% theoretical peak FLOPS on 512 GPUs (<xref ref-type="bibr" rid="B60">Shoeybi et al., 2019</xref>). Model parallelism is particularly beneficial in accelerating encoder layers of the transformer network because of its bidirectional properties. We utilize model parallelism in accelerating the encoder layers of the transformer network accelerator.</p>
</sec>
<sec id="s2-2-4">
<title>2.2.4 Sparse Attention</title>
<p>The SDPA mechanism described in <xref ref-type="sec" rid="s2-1-1">Section 2.1.1</xref> uses the full attention mechanism where the query vector is compared to all key and value vectors. The full attention mechanism is the most accurate among different attention patterns and is ideal for shorter sequences. However, because of the quadratic computational complexity of the full attention pattern, it may be necessary to use an attention pattern with less computational complexity at higher sequence lengths. Introducing sparsity in the attention layers has been one of the prominent algorithm optimizations to reduce the computational complexity of the full attention mechanism. Sparse attention only attends to the keys and values relevant to the query. Two major types of sparse attention have been proposed: locality-based sparsity and content-based sparsity. We use both types of sparsity in designing our transformer network accelerator.</p>
<p>
<italic>Locality-based sparsity</italic> uses fixed/random patterns based on the query&#x2019;s relative position to the current time step. The type of data determines the choice of attention pattern, and each attention pattern has its strengths and weaknesses. Longformer (<xref ref-type="bibr" rid="B1">Beltagy et al., 2020</xref>) proposed the sliding window and dilated sliding window attention patterns and achieved state-of-the-art results for character modeling tasks. The <italic>sliding window attention</italic>, obtained by focusing on sequence elements <italic>t</italic>
<sub>
<italic>sld</italic>
</sub> time steps away, is ideal for data with high spatial locality. The sliding window can also be dilated where it only pays attention to every other <italic>t</italic>
<sub>
<italic>sld</italic>
</sub> time step. Sparse transformers (<xref ref-type="bibr" rid="B8">Child et al., 2019</xref>) proposed <italic>strided attention</italic> and <italic>strided &#x2b; sliding window attention</italic> patterns and achieved equal or better accuracy compared to full attention mechanisms while reducing the number of operations. Sparse attention is used for music and image generation and machine translations of text. <italic>Strided attention</italic> is obtained by skipping <italic>t</italic>
<sub>
<italic>str</italic>
</sub> time steps and performs well in repeating or oscillating sequences such as audio or video data. A combination of <italic>strided global attention with sliding window attention</italic> is recommended for long documents (<xref ref-type="bibr" rid="B1">Beltagy et al., 2020</xref>) where there may be a correlation in texts that are farther away.</p>
<p>Content-based sparsity is another type of sparsity and is based on the similarity of key-value attention pairs with the query. <italic>Content-based sparsity</italic> uses the similarity of the current time step with the previous time step to determine the sparsity. The outing transformer (<xref ref-type="bibr" rid="B54">Roy et al., 2021</xref>) uses <italic>k</italic>-means clustering while Sinkhorn network (<xref ref-type="bibr" rid="B61">Tay et al., 2020a</xref>) uses sorting to determine sparsity. Reformer (<xref ref-type="bibr" rid="B27">Kitaev et al., 2020</xref>) uses angular locality-sensitive hashing (LSH) to accelerate the transformer for long sequences. Angular LSH uses random hyperplanes that pass through the origin, and the angle of a point is determined by its position with respect to the different hashing hyperplanes. By using LSH to hash, the complexity of the SDP attention is reduced from <italic>O</italic> (<italic>n</italic>
<sup>2</sup>) to <italic>O</italic> (<italic>n</italic>&#x2009;log&#x2009; <italic>n</italic>). We use angular LSH to introduce content-based sparsity in iMTransformer.</p>
<p>LSH is an approximate nearest neighbor search (NNS) approach for alleviating the curse of dimensionality when searching a large number of data points, thus, allowing for a fast NNS. This is accomplished by hashing similar items with the same binary signature. To perform an NNS, the binary signature of a query is compared to the binary signatures of the keys using a Hamming distance. LSH attention has been used in multiple attention-based neural network architectures (<xref ref-type="bibr" rid="B20">Kaiser et al., 2016</xref>; <xref ref-type="bibr" rid="B27">Kitaev et al., 2020</xref>).</p>
</sec>
</sec>
<sec id="s2-3">
<title>2.3 In-Memory Computing Based Hardware Kernels for Acceleration</title>
<p>To accelerate transformer networks, iMTransformer leverages crossbars to implement the SDPA and CAMs to realize content-based sparsity using LSH. Crossbars and CAMs are described in detail in <xref ref-type="sec" rid="s2-3-1">Section 2.3.1</xref>. Crossbars and CAMs can be implemented in CMOS or non-volatile devices based on emerging technologies such as FeFET. These devices are discussed in <xref ref-type="sec" rid="s2-3-2">Section 2.3.2</xref>. Finally, we review attention and transformer network accelerators in <xref ref-type="sec" rid="s2-3-3">Section 2.3.3</xref> based on crossbars and CAMs using CMOS and FeFET devices.</p>
<sec id="s2-3-1">
<title>2.3.1 Circuits</title>
<p>The transformer network attention mechanism uses linear layers and SDPA, requiring matrix-vector multiplications. Crossbars can accelerate matrix-vector multiplications, and have been used in a variety of DNN applications such as CNNs (<xref ref-type="bibr" rid="B57">Shafiee et al., 2016</xref>; <xref ref-type="bibr" rid="B6">Chen P.-Y. et al., 2018</xref>) and MANN (<xref ref-type="bibr" rid="B49">Ranjan et al., 2019</xref>). A <italic>crossbar</italic> (<xref ref-type="bibr" rid="B15">Gokmen and Vlasov, 2016</xref>) is an array-like circuit structure where each input is connected to every output, and vice versa as shown in <xref ref-type="fig" rid="F2">Figure 2A</xref>. To perform matrix-vector multiplications, the matrix must be encoded as conductances stored in the crossbar crosspoints <italic>g</italic>
<sub>
<italic>i</italic>,<italic>j</italic>
</sub>, and the vectors as input voltages <italic>V</italic>
<sub>
<italic>i</italic>
</sub>. The output of the matrix-vector multiplication is read as currents <italic>I</italic>
<sub>
<italic>j</italic>
</sub> at the columns using analog-to-digital converters (ADCs). The ADCs, typically, consume the majority of the energy (58%) and area (81%) of crossbar arrays (<xref ref-type="bibr" rid="B55">Roy et al., 2020</xref>). This energy and area consumption also increases exponentially with increased precision (<xref ref-type="bibr" rid="B57">Shafiee et al., 2016</xref>). NeuroSim, a DNN simulator (<xref ref-type="bibr" rid="B6">Chen P.-Y. et al., 2018</xref>), uses crossbars to benchmark convolutional and multilayer perceptron-based neural networks.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>
<bold>(A)</bold> Crossbars are array-like circuit structures where each input is connected to every output typically by a resistor at their crosspoints. <bold>(B)</bold> CAMs can perform fast parallel searches in their memory. <bold>(C)</bold> FeFET devices are CMOS compatible devices that provide non-volatility, high density, and low power consumption.</p>
</caption>
<graphic xlink:href="felec-03-847069-g002.tif"/>
</fig>
<p>
<italic>Content-addressable memories</italic> (CAMs) have been used to accelerate attention mechanisms for MANNs (<xref ref-type="bibr" rid="B44">Ni et al., 2019</xref>; <xref ref-type="bibr" rid="B30">Laguna A. F. et al., 2019</xref>). CAMs are a special type of memory that can perform fast parallel searches across the entire memory (<xref ref-type="fig" rid="F2">Figure 2B</xref>). Different CAM designs have been proposed for accelerating various search operations. <italic>Binary CAMs</italic> (BCAMs) store either a logic &#x201c;0&#x201d; or logic &#x201c;1&#x201d; in each cell while <italic>Ternary CAMs</italic> (TCAMs) can store an additional don&#x2019;t care value &#x201c;X,&#x201d; to signify that the bit can match to either a logic &#x201c;0&#x201d; or a logic &#x201c;1.&#x201d; Hamming distance is the most straightforward metric for approximate search in BCAMs/TCAMs. CAMs, which are traditionally used in routers and caches (<xref ref-type="bibr" rid="B23">Karam et al., 2015</xref>; <xref ref-type="bibr" rid="B70">Yin et al., 2020</xref>), have been gaining popularity in data-intensive applications such as nearest neighbor search (<xref ref-type="bibr" rid="B28">Kohonen, 2012</xref>; <xref ref-type="bibr" rid="B25">Kazemi et al., 2020</xref>, <xref ref-type="bibr" rid="B26">2021b</xref>), bioinformatics (<xref ref-type="bibr" rid="B29">Laguna et al., 2020</xref>), neural networks (<xref ref-type="bibr" rid="B69">Chang, 2009</xref>; <xref ref-type="bibr" rid="B9">Wang et al., 2010</xref>; <xref ref-type="bibr" rid="B35">Li C. et al., 2020</xref>, <xref ref-type="bibr" rid="B36">Li et al., 2021 H.</xref>; <xref ref-type="bibr" rid="B24">Kazemi et al., 2021a</xref>), etc. CAMs also have been used in implementing LSH-based attention (<xref ref-type="bibr" rid="B44">Ni et al., 2019</xref>).</p>
</sec>
<sec id="s2-3-2">
<title>2.3.2 Devices</title>
<p>Crossbars and CAMs can be implemented using CMOS or non-volatile memories (NVMs) based on emerging technologies such as resistive RAMs (ReRAMs) and FeFETS. CMOS-based crossbars and CAMs have low write latency and energy, making them ideal for transformer sublayers requiring many write operations. However, CMOS-based circuits have high leakage power (<xref ref-type="bibr" rid="B18">Jerry et al., 2018</xref>) which is detrimental for storing static weights.</p>
<p>Unlike CMOS-based memories, NVMs, such as ReRAM and FeFET devices, can store static weights without periodic refreshes. Moreover, NVMs also have a higher density compared to CMOS-based memories. ReRAMs offer good density and fast reads, making them a good candidate for applications requiring a large number of read operations. However, ReRAMs can suffer from cycle-to-cycle (C2C) variation and small <italic>G</italic>
<sub>max</sub>/<italic>G</italic>
<sub>min</sub> ratios (<xref ref-type="bibr" rid="B18">Jerry et al., 2018</xref>). Moreover, ReRAMs have low endurance (10<sup>6</sup>) compared to CMOS-based devices (10<sup>16</sup>) putting them at a disadvantage for write operations (<xref ref-type="bibr" rid="B73">Yu and Chen, 2016</xref>).</p>
<p>FeFET based crossbars and CAMs are also great candidates for high-read operations because of their high density and fast read operation. FeFETs have acceptable <italic>G</italic>
<sub>max</sub>/<italic>G</italic>
<sub>min</sub> and observe lower C2C variations compared to ReRAMs (<xref ref-type="bibr" rid="B18">Jerry et al., 2018</xref>). As shown in <xref ref-type="fig" rid="F2">Figure 2C</xref>, FeFETs have a similar structure to the metal-oxide-semiconductor field-effect transistors (MOSFETs) in standard CMOS. The only difference between the two is the extra layer of FE oxide deposited in the FeFET&#x2019;s gate stack. Because of this similarity, FeFETs can be integrated into the CMOS fabrication process (<xref ref-type="bibr" rid="B2">Beyer et al., 2020</xref>). This enables us to consider a combination of FeFETs and CMOS devices in our architecture to realize better performance. One of the shortcomings of FeFETs (as well as other emerging devices) is device variation. In order to alleviate device variation, we need to use write-verify programming schemes (<xref ref-type="bibr" rid="B58">Sharifi et al., 2021</xref>) which increases the write time. FeFET devices also have lower endurance (10<sup>10</sup>) than CMOS devices (<xref ref-type="bibr" rid="B73">Yu and Chen, 2016</xref>). This makes the use of FeFETs challenging for applications that demand a high number of write operations. Transformer networks also may require large-scale memories. Large-scale FeFET memories have been demonstrated (<xref ref-type="bibr" rid="B2">Beyer et al., 2020</xref>).</p>
</sec>
<sec id="s2-3-3">
<title>2.3.3 Attention and Transformer Network Accelerators</title>
<p>Different attention-based accelerators have previously been proposed utilizing CAMs (<xref ref-type="bibr" rid="B31">Laguna A. et al., 2019</xref>; <xref ref-type="bibr" rid="B5">Challapalle et al., 2020</xref>; <xref ref-type="bibr" rid="B30">Laguna A. F. et al., 2019</xref>), crossbar arrays (<xref ref-type="bibr" rid="B5">Challapalle et al., 2020</xref>; <xref ref-type="bibr" rid="B49">Ranjan et al., 2019</xref>), and GP-CIMs (<xref ref-type="bibr" rid="B50">Reis et al., 2020a</xref>). These accelerators were used to accelerate the attention mechanism for MANNs and RNNs. However, the attention mechanism for transformer networks has different properties than the ones in MANNs and RNNs. Existing attention accelerators (<xref ref-type="bibr" rid="B31">Laguna et al., 2019a</xref>,<xref ref-type="bibr" rid="B30">b</xref>; <xref ref-type="bibr" rid="B50">Reis et al., 2020a</xref>) use <italic>k</italic>-nearest neighbors, which are not applicable for the transformer network since the attentional weights need to be passed to the next layer. The transformer network also uses MHA, which can be highly parallelized and has not been utilized in existing attention accelerators. An increase in parallelism can also be achieved by exploiting the autoregressive and bidirectional properties of the MHA. These properties have not been considered in existing attention accelerators.</p>
<p>A ReRAM-based transformer (ReTransformer) (<xref ref-type="bibr" rid="B67">Yang X. et al., 2020</xref>) has been proposed to accelerate SDPA using ReRAM-based crossbars. ReTransformer uses matrix decomposition to avoid writing the intermediate results. ReRAM has lower endurance compared to CMOS and FeFETs; hence writing in the crossbars must be minimized (<xref ref-type="bibr" rid="B73">Yu and Chen, 2016</xref>). However, attention caching (<xref ref-type="bibr" rid="B11">Dai et al., 2019</xref>), which necessitates writing in the crossbars, has been shown to improve the execution time of transformer networks for long sequences by reducing the number of operations. The low endurance of ReRAMs makes them undesirable as memory devices for attentional caches. Compared to the GPU approach, ReTransformer achieves a speedup of 23.21&#xd7; with a 1086&#xd7; power reduction.</p>
</sec>
</sec>
</sec>
<sec id="s3">
<title>3 Methodology</title>
<p>To eliminate the limitation from memory bandwidth and exploit transformer networks&#x2019; high degree of achievable parallelism as sequence length increases, we propose an in-memory computing-based transformer network architecture referred to as iMTransformer. iMTransformer follows a hierarchical design style. <xref ref-type="sec" rid="s3-1">Section 3.1</xref> presents an overview of iMTransformer. Transformer networks have three types of MHA: bidirectional, masked, and encoder-decoder. Each of these MHA types has different characteristics that can be exploited to improve the latency and energy performance of iMTransformer. The mapping of attention mechanisms for different types of MHA is expounded in <xref ref-type="sec" rid="s3-2">Section 3.2</xref>. The computational complexity of transformer networks can be reduced by introducing either a content-based or locality-based sparsity. <xref ref-type="sec" rid="s3-3">Section 3.3</xref> aims to reduce energy consumption by utilizing sparsity. Energy can also be improved by utilizing FeFETs for sublayers with high read rates and CMOS for sublayers with high write rates. This technology-level mapping is explained in <xref ref-type="sec" rid="s3-4">Section 3.4</xref>. Finally, we summarize the circuit and device level mapping in <xref ref-type="sec" rid="s3-5">Section 3.5</xref>.</p>
<sec id="s3-1">
<title>3.1 In-Memory Transformer Network Architecture</title>
<p>iMTransformer is organized in a hierarchical pattern of banks, tiles, and mats as shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. This hierarchical pattern follows existing memory hierarchies and the hierarchical pattern of transformer networks as shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. The transformer network is composed of encoder layers and/or decoder layers. Each encoder or decoder layer is composed of MHA, FF, and normalization layers. The MHA then can be greatly parallelized into multiple attention heads.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>The iMTransformer follows a hierarchical memory structure. The iMTransformer <bold>(A)</bold> is composed of encoder <bold>(B)</bold> and decoder <bold>(C)</bold> banks. The encoder bank is composed of an encoder MHA tile <bold>(D)</bold> and FF tiles and normalization units. The decoder bank is composed of two decoder MHA tiles <bold>(E)</bold>, an FF tile and a normalization unit. The encoder <bold>(D)</bold> and decoder <bold>(E)</bold> MHA tiles are composed of AH Mats <bold>(F)</bold> and an aggregator unit.</p>
</caption>
<graphic xlink:href="felec-03-847069-g003.tif"/>
</fig>
<p>The encoder and decoder layers of transformer networks have different properties which can be exploited to increase parallelism and reduce the number of computations in implementing transformer networks. Hence, iMTransformer uses two types of banks: encoder banks and decoder banks. For an encoder-decoder transformer network, the information first flows through the encoder banks (i.e., Enc Banks 1&#x2013;6 in <xref ref-type="fig" rid="F3">Figure 3A</xref>) before being processed by the decoder banks (i.e., Dec Banks 1&#x2013;6 in <xref ref-type="fig" rid="F3">Figure 3A</xref>). Enc Banks processes data one after another. The output of the final Enc Bank is then passed to all the Dec Banks as the stored key-value pairs of the SDPA. These key-value pairs are then processed in parallel. Because the decoder layers are autoregressive, the input query of Dec Bank 1 is the output of Dec Bank 6.</p>
<p>As shown in <xref ref-type="fig" rid="F1">Figures 1C,D</xref>, each encoder and decoder layer of the transformer network consists of multi-head attention, feedforward, and normalization sublayers. The iMTransformer banks (<xref ref-type="fig" rid="F3">Figures 3B,C</xref>) are hence composed of a normalization unit (NU), the feedforward tile (FF Tile), and the multi-head attention memory tile (MHA Tile). The NUs execute the layer normalization using additions and shift operations. Since no weights are stored in the NUs, they are shared between sublayers. The FF Tile is composed of two crossbar sub-arrays and performs feedforward layer operations with ReLu activation in between. Encoder banks (<xref ref-type="fig" rid="F3">Figure 3B</xref>) have one MHA tile while decoder banks (<xref ref-type="fig" rid="F3">Figure 3C</xref>) have two MHA tiles. To follow the information flow in <xref ref-type="fig" rid="F1">Figure 1C</xref>, the encoder bank (<xref ref-type="fig" rid="F3">Figure 3B</xref>) passes the information to the MHA tile and then to the NU. Then, the NU passes the information to the FF tile and then back to the NU for the output. To map the decoder layer in <xref ref-type="fig" rid="F1">Figure 1D</xref> to iMTransformer, the decoder bank&#x2019;s input (<xref ref-type="fig" rid="F3">Figure 3E</xref>) is passed to one of its MHA Tiles &#x24b6;. The attention vector from the MHA tile is then passed to the NU &#x24b7;. The normalized attention vector from the NU is then passed to a different MHA tile &#x24b8;and back to the NU &#x24b9;and then to the FF Tile &#x24ba;. The final attention vector is then passed to the NU for output &#x24bb;.</p>
<p>The MHA (<xref ref-type="fig" rid="F1">Figure 1B</xref>) splits the input into queries, keys, and values and passes them to different attention heads. The different attention heads are then concatenated together using a linear function. Since each attention head can be executed in parallel with each other, we decompose each attention head into multiple attention memory mats &#x278a; (AH Mat), where each mat represents an attention head. The encoder MHA tile (<xref ref-type="fig" rid="F3">Figure 3D</xref>) also groups multiple AH mats to take advantage of the bidirectional property of encoder MHA. The details for this are further discussed in <xref ref-type="sec" rid="s3-2">Section 3.2</xref>. Afterward, the aggregator unit (AU) in the MHA Tile (<xref ref-type="fig" rid="F3">Figure 3D</xref> &#x278b;) computes a linear combination of the different attention heads.</p>
<p>Each attention head of the MHA is mapped onto an AH Mat. The AH Mat (<xref ref-type="fig" rid="F3">Figure 3E</xref>) is composed of three projection units (PU), two caching units (CU), a hashing unit (HU), a sparsity unit (SU), an attention selector (AS), and a softmax lookup table (LUT). As shown in <xref ref-type="fig" rid="F1">Figure 1B</xref>, each MHA head is composed of three linear layers that project the query, key, and value into a different subspace. These linear layers are implemented by PU-Q, PU-K, and PU-V, which store the projection matrices of the linear layers (<bold>W</bold>
<sup>
<bold>Q</bold>
</sup>, <bold>W</bold>
<sup>
<bold>K</bold>
</sup>, <bold>W</bold>
<sup>
<bold>V</bold>
</sup>) to project the input to different feature spaces. These projection matrices have static weights during inference. The projected query <inline-formula id="inf2">
<mml:math id="m3">
<mml:mi>q</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">Q</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msubsup>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula>, obtain from PU-Q, is then compared to the present and previous projected keys <bold>K</bold> &#x3d; {<italic>k</italic>
<sub>
<italic>t</italic>
</sub>, <italic>k</italic>
<sub>
<italic>t</italic>&#x2212;1</sub>, <italic>k</italic>
<sub>
<italic>t</italic>&#x2212;2</sub> &#x2026; }. and values <bold>V</bold> &#x3d; {<italic>v</italic>
<sub>
<italic>t</italic>
</sub>, <italic>v</italic>
<sub>
<italic>t</italic>&#x2212;1</sub>, <italic>v</italic>
<sub>
<italic>t</italic>&#x2212;2</sub> &#x2026; } obtained from PU-K <inline-formula id="inf3">
<mml:math id="m4">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">K</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msubsup>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and PU-V <inline-formula id="inf4">
<mml:math id="m5">
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">V</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msubsup>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> respectively. &#x2780; The PUs (shown in green boxes as in <xref ref-type="fig" rid="F3">Figure 3F</xref>) process the input in parallel. The output of the PUs represent the input to the SDPA, which are implemented using the CUs and the softmax LUT (shown as dark blue boxes in <xref ref-type="fig" rid="F3">Figure 3F</xref> &#x2781;&#x2013;&#x2785;).</p>
<p>The SDPA is mapped to the CUs and the softmax LUT. Specifically, the keys and values are cached in crossbars (CU-K and CU-V, respectively) for reuse to reduce computation &#x2781;. The outputs of PU-K <inline-formula id="inf5">
<mml:math id="m6">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">K</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msubsup>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and PU-V <inline-formula id="inf6">
<mml:math id="m7">
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">V</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msubsup>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> are written to a column of CU-K and a row of CU-V, respectively. During inference, CU-K and CU-V are constantly written to, while PU-Q, PU-K, PU-V have static weights. &#x2782; The output by PU-Q <italic>q</italic> is used as the input to CU-K. &#x2783; The output of CU-K, <italic>q</italic>
<bold>K</bold>
<sup>
<italic>T</italic>
</sup>, passes through the softmax LUT. The dot product scaling <inline-formula id="inf7">
<mml:math id="m8">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is implemented by dropping the three least significant bits for a <italic>d</italic>
<sub>
<italic>k</italic>
</sub> &#x3d; 64. This is equivalent to dividing 8. [<italic>d</italic>
<sub>
<italic>k</italic>
</sub> &#x3d; 64 is used in most transformer network (<xref ref-type="bibr" rid="B63">Vaswani et al., 2017</xref>; <xref ref-type="bibr" rid="B12">Devlin et al., 2018</xref>)]. &#x2784; The output of the softmax unit is then used as the input to CU-V. &#x2785; CU-V produces the final output of the AH Mat. To combine the outputs of AH Mats, the AU concatenates and pools the output of each AH Mat in an MHA Tile. The AU stores the weights <bold>W</bold>
<sup>
<bold>O</bold>
</sup> which are static during inference, and &#x278b; outputs the final multi-head attention score.</p>
<p>The iMTransformer stores a pre-trained transformer network model. Hence, the sizes and number of crossbar arrays for the PUs and HU can be set to exactly store the trained weights. However, the sequence length of each input is variable. Hence the size of the crossbars in the CUs and CAMs in the SU cannot be fixed. Since the keys are stored column-wise, increasing the number of keys (increasing the sequence length) will only require crossbar arrays that are implemented in parallel. On the other hand, as the sequence length increases, the values will require more rows. This will also require more crossbar arrays where each crossbar array outputs a partial sum that needs to be aggregated. We limit the sequence length to 512 to prevent a large number of partial sums and use content-based sparsity for sequence length <italic>n</italic> &#x3e; 4096. The CAMs are also designed to hold up to sequence lengths up to <italic>n</italic> &#x3d; 4096. Beyond this, a replacement algorithm must be designed to determine which sequence elements can be removed from the memory. We have not explored this replacement algorithm in this research, and it is a topic for future work.</p>
</sec>
<sec id="s3-2">
<title>3.2 Multi-Head Attention Operation Mapping</title>
<p>As discussed in <xref ref-type="sec" rid="s2-1-3">Section 2.1.3</xref>, transformer networks are heavily dominated by MHAs. Hence iMTransformer focuses on accelerating MHA, which relies primarily on matrix-vector multiplications. Transformer networks have three different types of attention: bidirectional MHA, masked MHA, and encoder-decoder MHA. Each type has different properties, which are exploited by iMTransformer to improve the time and energy consumption of transformer networks. The bidirectional MHA can be parallelized during inference as the entire input sequence is available and does not need to be computed. On the other hand, the energy consumption of the masked MHA can be reduced by turning off columns in the CUs. Finally, encoder-decoder MHA can parallelize the computation of the projected keys and values across different layers. In this section, we discuss the operation mapping for the masked MHA (<xref ref-type="sec" rid="s3-2-1">Section 3.2.1</xref>), bidirectional MHA (<xref ref-type="sec" rid="s3-2-2">Section 3.2.2</xref>) and the encoder-decoder MHA in <xref ref-type="sec" rid="s3-2-3">Section 3.2.3</xref>.</p>
<sec id="s3-2-1">
<title>3.2.1 Masked Multi-Head Attention</title>
<p>The masked MHA uses a decoder MHA tile and does not use model parallelism (<xref ref-type="fig" rid="F4">Figure 4A</xref>). The transformer&#x2019;s decoder layer is autoregressive by nature (i.e., the input is a delayed version of the output) and uses masked MHA. The masking is essential to prevent a backward information flow (i.e., the future affects the past). The masked MHA computes the projection layers and SDPA for each sequence element before computing the next sequence element.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>The timeline of implementation for the different types of MHA namely masked MHA <bold>(A)</bold>, bi-drectional MHA <bold>(B)</bold> and encoder-decoder MHA <bold>(C)</bold>. The bi-directional MHA uses model parallelism across multiple AHUs while the encoder-decoder MHA parallelizes the computation of the keys and values across different decoder banks.</p>
</caption>
<graphic xlink:href="felec-03-847069-g004.tif"/>
</fig>
<p>We utilize an AS, a SU, and a HU to implement masking and sparsity (i.e., locality-based sparsity and content-based sparsity). When masking is implemented, the AS disables the rows of the CU-V that represent future time steps when selecting the values for the SDPA. The AS is a configurable circular shift register that allows different types of locality-based sparsity based on the stored pattern in the register. More details regarding locality-based sparsity and the different attention patterns are discussed in <xref ref-type="sec" rid="s3-3-1">Section 3.3.1</xref>. The HU is a crossbar array that hashes the keys using LSH based on random projection and stores the hash signatures <italic>H</italic>(<italic>K</italic>) in the SU. Content-based sparsity mapping is further expounded in <xref ref-type="sec" rid="s3-3-2">Section 3.3.2</xref>.</p>
</sec>
<sec id="s3-2-2">
<title>3.2.2 Bidirectional Multi-Head Attention</title>
<p>The encoder layer uses bidirectional MHA to attend to each sequence element&#x2019;s previous, current, and future values. The bidirectional MHA can be parallelized by duplicating weights in the PUs. <xref ref-type="fig" rid="F4">Figure 4B</xref> shows the operations for each time step. (Step 1) The PU-K and PU-V of all AH Mats in the same tile perform matrix-vector multiplications in parallel. (Step 2) The first AH Mat then broadcasts the computed key-value pair (<italic>k</italic>
<sub>1</sub>, <italic>v</italic>
<sub>1</sub>) to other AH Mats. (Step 3) Each AH Mat then writes (<italic>k</italic>
<sub>1</sub>, <italic>v</italic>
<sub>1</sub>) to the CU-K column-wise and CU-V row-wise. Steps 2-3 are repeated for (<italic>k</italic>
<sub>2</sub>, <italic>v</italic>
<sub>2</sub>) and so on, until the whole sequence is processed. For a sequence length of <italic>n</italic> and the degree of parallelism <italic>p</italic> (Step 2n/p to 2n/p&#x2b;5), each AH Mat performs the SDPA operation (<italic>q</italic>
<bold>K</bold>
<sup>
<italic>T</italic>
</sup>) for each sequence element. However, the MHA still has an <italic>O</italic>(<italic>n</italic>) complexity because of the number of writes. The complexity of the SDPA is reduced to <italic>O</italic>(<italic>n</italic>) for <italic>n</italic> &#x2264; <italic>p</italic> (<italic>p</italic> &#x3d; 3 in <xref ref-type="fig" rid="F3">Figure 3</xref>). If <italic>n</italic> &#x3e; <italic>p</italic>, the time complexity of the SDPA is <italic>O</italic> (<italic>n</italic>/<italic>p</italic>). However, the increased parallelism requires more AH Mats and thus a higher peak power. As such, the attention parallelism is limited by the total memory size and/or the thermal design power.</p>
</sec>
<sec id="s3-2-3">
<title>3.2.3 Encoder-Decoder MHA</title>
<p>The encoder-decoder MHA (<xref ref-type="fig" rid="F1">Figure 1D</xref>) is implemented by the a decoder MHA tile &#x24b8; in a Dec Bank (<xref ref-type="fig" rid="F1">Figure 1C</xref>). The key-value pair inputs of the MHA tiles (<bold>K</bold>&#x2032;, <bold>V</bold>&#x2032;) all come from the output of the last encoder bank (Enc Bank 6). Hence the encoder-decoder MHA requires communication between the encoder and the decoder banks. Instead of serializing the computation of the keys and values per decoder layer as in the standard iMTransformer implementation discussed in <xref ref-type="sec" rid="s3-1">Section 3.1</xref>, the computation of the projected key-value pairs (<bold>K</bold>, <bold>V</bold>) can be parallelized for all decoder banks (Dec 1&#x2013;6) in <xref ref-type="fig" rid="F3">Figure 3A</xref>. The encoder-decoder MHA processes the decoder sequence query one element at a time as shown in <xref ref-type="fig" rid="F4">Figure 4C</xref>. However, the keys and values do not need to be recalculated.</p>
</sec>
</sec>
<sec id="s3-3">
<title>3.3 Sparse Attention for iMTransformer</title>
<p>A full self-attention mechanism (<xref ref-type="fig" rid="F5">Figure 5A</xref>) computes for the global correlation between elements in a sequence. However, computing for the full self-attention mechanism can be costly as the sequence length increases (<xref ref-type="bibr" rid="B8">Child et al., 2019</xref>). Introducing sparsity in the attention mechanism can reduce the transformer network&#x2019;s computational complexity without sacrificing accuracy. ADCs in a crossbar array consume most of the energy in performing matrix-vector multiplications (<xref ref-type="bibr" rid="B55">Roy et al., 2020</xref>). By introducing sparsity, the ADCs in a crossbar that does not contribute to improving the accuracy can be turned off to reduce energy consumption. As discussed in <xref ref-type="sec" rid="s2-2-4">Section 2.2.4</xref>, there are two main types of sparse attention: locality-based sparse attention and content-based sparse attention. Locality-based sparse attention (<xref ref-type="sec" rid="s3-3-1">Section 3.3.1</xref>) focuses on the temporal locality between sequence elements while content-based sparse attention (<xref ref-type="sec" rid="s3-3-2">Section 3.3.2</xref>) focuses on the similarity between sequence elements.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>The full attention <bold>(A)</bold> and masked attention pattern <bold>(B)</bold> have high computational complexity. To improve the computational efficiency, different locality-based sparse attention patterns can be used to improve the computational efficiency. Examples of these attention patterns are strided <bold>(C)</bold>, sliding window <bold>(D)</bold>, dilated sliding window <bold>(E)</bold> and strided <bold>(F)</bold> sliding window attention. The configurable attention selector <bold>(G)</bold> uses a circular shift register which contains a pre-defined attention pattern based on the type of attention matrix used.</p>
</caption>
<graphic xlink:href="felec-03-847069-g005.tif"/>
</fig>
<sec id="s3-3-1">
<title>3.3.1 Locality-Based Sparse Attention</title>
<p>Locality-based sparse attention focuses on the sequence elements based on their position relative to the query. Different attention patterns have been proposed that allow more efficient computation of attention for longer sequences without sacrificing accuracy. Examples of supported attention patterns include: strided attention (<xref ref-type="fig" rid="F5">Figure 5C</xref>), sliding window attention (<xref ref-type="fig" rid="F5">Figure 5D</xref>), dilated sliding window attention (<xref ref-type="fig" rid="F5">Figure 5E</xref>) and strided sliding window attention (<xref ref-type="fig" rid="F5">Figure 5F</xref>). It is also possible to combine masking with any of these attention patterns.</p>
<p>iMTransformer uses a configurable AS (as shown in <xref ref-type="fig" rid="F5">Figure 5G</xref>), which consists of a circular shift register to store a predefined attention pattern for determining the crossbar columns to be activated. The shift register is configurable to implement different sparsity patterns. We use a 128-bit circular shift register to control a 64-column crossbar (<xref ref-type="fig" rid="F5">Figure 5</xref>). The first 64 bits determine the crossbar columns to be activated. The last 64 bits function as a buffer necessary to implement certain attention patterns and masking. The shift register is shifted to the right by 1&#xa0;bit at every time step. The stored pattern in the AS is different for each attention pattern, as shown in <xref ref-type="fig" rid="F5">Figure 5G</xref>. The AS has all 1&#x2019;s in its register for the full attention pattern. The strided attention activates every other <italic>c</italic> column. For the 128-bit circular shift register, <italic>c</italic> must be a factor of 128. The sliding window width can be configured by setting the number of 1&#x2019;s stored in the AS.</p>
</sec>
<sec id="s3-3-2">
<title>3.3.2 Content-Based Sparse Attention</title>
<p>Another type of sparsity is based on the similarity of the query with the keys, or content-based sparsity. Content-based sparsity can be implemented using LSH. LSH reduces the number of attention computations for CU-K and CU-V. We implemented angular LSH using random hyperplanes to represent cosine distance. To implement an LSH-based attention mechanism, the keys need to be hashed to a binary signature, where each bit in the signature is hashed using equation <italic>H</italic>(<italic>q</italic>) &#x3d; (sign (<italic>q</italic> &#x22c5; <italic>r</italic>) &#x2b; 1)/2. The logic &#x201c;0&#x201d; or &#x201c;1&#x201d; determines if the point is in the left or right of the hyperplane, respectively.</p>
<p>In iMTransformer, the hash function <italic>H</italic>(<italic>q</italic>) is implemented by the HU using crossbars. The binary signature of the query <italic>H</italic>(<italic>q</italic>) is compared to the signature of keys <italic>H</italic>(<italic>K</italic>), using a CAM-based SU. The SDPA, implemented by the CU-K and CU-V, is only implemented on the keys with <italic>m</italic> most similar buckets as <italic>H</italic>(<italic>q</italic>). By only focusing on the most similar keys and values with the query, we reduce the required multiplications that are more computationally expensive. The hashing function introduces an additional latency and energy overhead when computing the signature. However, it can reduce the softmax computations and value comparisons as the sequence length increases. Thus, LSH-based sparsity is only beneficial when the computational savings associated with avoiding comparisons with all key-value pairs exceeds the hashing overhead. A study of the effect of the hashing overhead and the effect of increasing sequence length is later explained in <xref ref-type="sec" rid="s4-3-4">Section 4.3.4</xref>.</p>
</sec>
</sec>
<sec id="s3-4">
<title>3.4 Device-Level Mapping</title>
<p>Different devices have different properties that make them ideal for certain operations. CMOS devices, for example, are ideal when the devices need to perform a large number of write operations because of their low write energy and high endurance. However, CMOS devices also have high leakage power, making them undesirable when weights must be stored for extended periods. Alternatively, FeFETs, are non-volatile and have low leakage power. However, FeFETs have much lower endurance compared to CMOS devices. Thus, CMOS devices are better for operations tasks that require a high number of write operations, while FeFETs are better when there are minimal writes and non-volatility is important.</p>
<p>To determine which devices are good fits for iMTransformer, we examine the usage of IMC kernels for iMTransformer. iMTransformer employs crossbars for two different reasons: (i) to store attentional (in PUs and AU) and feedforward weights (in FF Tiles), and (ii) to serve as attentional caches to store intermediate activation (in CUs). The PUs, AU, FF Tiles, and CUs perform different operations and require different properties. The weights in PU, AU, and FF Tiles, once trained, do not change. Hence, they do not require additional write operations and would benefit from using NVMs such as FeFET devices. Alternatively, CUs require frequent write operations and would benefit from memories with lower write times and energy and higher endurance, such as CMOS devices. Therefore, iMTransformer employs FeFET-based crossbars for attentional and feedforward weights because of the FeFET&#x2019;s non-volatility and CMOS-based crossbars as attentional caches because of the CMOS&#x2019;s low write energy and high endurance. The FF Tiles and AU also use FeFET-based crossbars as they also do not require writes and can benefit from FeFET&#x2019;s non-volatility.</p>
<p>FeFET-based crossbars have been proposed in <xref ref-type="bibr" rid="B7">Chen X. et al. (2018)</xref>. However, this crossbar design stores binary weights and performs XNOR operations. To utilize FeFET-based crossbars, we use a binary-code-based quantization technique introduced for transformer networks (<xref ref-type="bibr" rid="B10">Chung et al., 2020</xref>). The binary-code-based quantization uses non-uniform quantization and decouples a feature vector into a scaling factor and a binary vector. Scaling can be done in the crossbar&#x2019;s ADCs, and only bitwise XNOR operations are necessary.</p>
</sec>
<sec id="s3-5">
<title>3.5 Summary of Mapping</title>
<p>To summarize the mapping of the iMTransformer architecture, <xref ref-type="table" rid="T1">Table 1</xref> shows the functional units of iMTransformer and their mapping to circuits and devices. The PUs, HUs, AUs, and FFUs are implemented using FeFET-based crossbars, while CUs are implemented using CMOS-based crossbars. On the other hand, the SU is implemented using CMOS-based CAM, and AS is implemented using a CMOS-based shifter.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Summary of hardware mapping of Transformer Network to iMTransformer.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Transformer network</th>
<th align="center">iMTransformer</th>
<th align="center">Crossbar</th>
<th align="center">CAM</th>
<th align="center">Shifter</th>
<th align="center">CMOS</th>
<th align="center">FeFET</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Linear Unit</td>
<td align="left">PU-Q</td>
<td align="center">
<italic>&#x2713;</italic>
</td>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="center">
<italic>&#x2713;</italic>
</td>
</tr>
<tr>
<td align="left">Linear Unit</td>
<td align="left">PU-K</td>
<td align="center">
<italic>&#x2713;</italic>
</td>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="center">
<italic>&#x2713;</italic>
</td>
</tr>
<tr>
<td align="left">Linear Unit</td>
<td align="left">PU-V</td>
<td align="center">
<italic>&#x2713;</italic>
</td>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="center">
<italic>&#x2713;</italic>
</td>
</tr>
<tr>
<td align="left">Attention Cache</td>
<td align="left">CU-V</td>
<td align="center">
<italic>&#x2713;</italic>
</td>
<td align="left"/>
<td align="left"/>
<td align="center">
<italic>&#x2713;</italic>
</td>
<td align="left"/>
</tr>
<tr>
<td align="left">Attention Cache</td>
<td align="left">CU-K</td>
<td align="center">
<italic>&#x2713;</italic>
</td>
<td align="left"/>
<td align="left"/>
<td align="center">
<italic>&#x2713;</italic>
</td>
<td align="left"/>
</tr>
<tr>
<td align="left">Hash Function</td>
<td align="left">HU</td>
<td align="center">
<italic>&#x2713;</italic>
</td>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="center">
<italic>&#x2713;</italic>
</td>
</tr>
<tr>
<td align="left">Hash Table</td>
<td align="left">SU</td>
<td align="left"/>
<td align="center">
<italic>&#x2713;</italic>
</td>
<td align="left"/>
<td align="center">
<italic>&#x2713;</italic>
</td>
<td align="left"/>
</tr>
<tr>
<td align="left">Sparsity</td>
<td align="left">AS</td>
<td align="left"/>
<td align="left"/>
<td align="center">
<italic>&#x2713;</italic>
</td>
<td align="center">
<italic>&#x2713;</italic>
</td>
<td align="left"/>
</tr>
<tr>
<td align="left">Linear Layer</td>
<td align="left">AU</td>
<td align="center">
<italic>&#x2713;</italic>
</td>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="center">
<italic>&#x2713;</italic>
</td>
</tr>
<tr>
<td align="left">Feedforward Layer</td>
<td align="left">FFU</td>
<td align="center">
<italic>&#x2713;</italic>
</td>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="center">
<italic>&#x2713;</italic>
</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s4">
<title>4 Results and Evaluation</title>
<p>This section discusses the evaluation of executing transformer networks using iMTransformer. We follow a bottom-up approach beginning with an array-level evaluation in <xref ref-type="sec" rid="s4-1">Section 4.1</xref> and experiment setup in <xref ref-type="sec" rid="s4-2">Section 4.2</xref>. The latency and energy evaluation of using MHA is then shown in <xref ref-type="sec" rid="s4-3">Section 4.3</xref>. For the evaluation in <xref ref-type="sec" rid="s4-3">Section 4.3</xref>, we only assume CMOS-based crossbars. An end-to-end accuracy, latency, and energy evaluation are then presented in <xref ref-type="sec" rid="s4-4">Section 4.4</xref>. We also show the effect of using CMOS-FeFET hybrid iMTransformer implementations where CMOS-based crossbars are used as attentional cache, and FeFET-based crossbars are used to store static weights in <xref ref-type="sec" rid="s4-4">Section 4.4</xref>. Finally, we compare iMTransformer using the MLPerf Inference benchmark for edge devices in <xref ref-type="sec" rid="s4-5">Section 4.5</xref>.</p>
<sec id="s4-1">
<title>4.1 Array-Level Evaluation</title>
<p>iMTransformer relies heavily on crossbars and CAMs in implementing transformer networks. We consider crossbars and CAMs implemented in a 14&#xa0;nm technology node. Both CMOS and FeFET devices are used for crossbars, while only CMOS devices are evaluated for CAMs. We used Neurosim to obtain the energy and delay results for reading and writing in crossbar arrays. For the 64 &#xd7; 64 crossbar arrays, each synapse has 8-bit precision (8 SRAM cells). Each crossbar has one 8-bit ADC per 8 columns, and their overhead is accounted for in the results. We obtained the CAM results using SPICE simulations based on the model used in <xref ref-type="bibr" rid="B72">Yin et al. (2017)</xref>. Based on our estimation, the interconnects between the arrays add about 20% overhead to latency and energy of the overall architecture. We calculated the interconnection delay by estimating the length of the longest wire in the interconnections. We used the 22&#xa0;nm design rules (wire width and pitch) to calculate the capacitance and resistance of the longest wire. Using the calculated numbers, we simulated the RC model for the wires and calculated the delay of the longest wire using SPICE simulations.</p>
<p>
<xref ref-type="table" rid="T2">Table 2</xref> shows the latency and energy results for 64 &#xd7; 64 CMOS-based and FeFET-based crossbars as well as the 64 &#xd7; 64 CMOS-based CAM array. The write latency and energy shown is a write for a single row, while the read latency and energy are for the whole array. CMOS-based crossbars have lower latencies than FeFET-based crossbars. Alternatively, FeFET-based crossbars have 4.12&#xd7; lower read energy than CMOS-based crossbars. CMOS-based crossbars have faster writes and lower write energy than FeFET-based crossbars. However, these array-level evaluations do not consider the leakage current necessary to store the weights in a CMOS-based crossbar. FeFET-based crossbars are superior for this figure of merit.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Latency and energy array-level results for CMOS and FeFET crossbars and CAMs.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="left"/>
<th align="center">Latency (ns)</th>
<th align="center">Energy (fJ)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="left">CMOS Crossbar</td>
<td align="left">Read (whole array)</td>
<td align="char" char=".">17.06</td>
<td align="char" char=".">78270</td>
</tr>
<tr>
<td align="left">Write (single row)</td>
<td align="char" char=".">1.04</td>
<td align="char" char=".">2.5</td>
</tr>
<tr>
<td rowspan="2" align="left">FeFET Crossbar</td>
<td align="left">Read (whole array)</td>
<td align="char" char=".">17.39</td>
<td align="char" char=".">1900</td>
</tr>
<tr>
<td align="left">Write (single row)</td>
<td align="char" char=".">175</td>
<td align="char" char=".">118</td>
</tr>
<tr>
<td rowspan="2" align="left">CMOS CAM</td>
<td align="left">Read (whole array)</td>
<td align="char" char=".">0.181</td>
<td align="char" char=".">441</td>
</tr>
<tr>
<td align="left">Write (single row)</td>
<td align="char" char=".">0.25</td>
<td align="char" char=".">1570</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4-2">
<title>4.2 Experiment Setup</title>
<p>We use the following as our baseline hardware for comparison: an Intel Core i7-10750H CPU (2.60GHz, 2592&#xa0;Mhz, six cores) with a Titan RTX GPU (672&#xa0;GB/s memory bandwidth and peak performance of 130 TFLOPS). To measure the latency and energy of the baseline implementation of the vanilla transformer (<xref ref-type="bibr" rid="B63">Vaswani et al., 2017</xref>), we use NVIDIA NSight and python line-profiler. We use <italic>nvidia-smi</italic> to collect the operating power while the transformer network runs. The average power is then multiplied by the average latency to obtain the energy. We use 8-bit quantization for both the GPU and iMTransformer using quantization-aware fine-tuning (<xref ref-type="bibr" rid="B74">Zafrir et al., 2019</xref>). 8-bit quantization is the lowest quantization for both weights and activations and achieves acceptable accuracies (<xref ref-type="bibr" rid="B74">Zafrir et al., 2019</xref>; <xref ref-type="bibr" rid="B39">Li et al., 2020c</xref>).</p>
<p>Using the python line-profiler, we obtain the number of operations executed in each stage and map them to the number of operations in iMTransformer. The total number of operations varies depending on the sequence length, the number of layers, the embedding size, and the number of heads in the MHA. We then calculate the latency and energy of iMTransformer from the number of operations and the corresponding array level results.</p>
<sec id="s4-2-1">
<title>4.2.1 Transformer Models and Datasets</title>
<p>In our experiments, we use three transformer models: the Vanilla transformer, the BERT-base, and BERT-large. The vanilla transformer has six encoder and six decoder layers. The vanilla transformer also has 512-dimensional embedding and eight attention heads in each MHA. The BERT-base has 12 encoder layers, with each layer having 12 heads in each MHA. In comparison, BERT-large has 24 encoder layers with 16 heads in each MHA. The multihead attention model used in <xref ref-type="sec" rid="s4-4-1">Section 4.4.1</xref> uses a 512-dimensional embedding (similar to the Vanilla Transformer). We use the Vanilla transformer and BERT-base parameters in <xref ref-type="sec" rid="s4-4-2">Section 4.4.2</xref>. The maximum sequence length is not restricted. BERT-base is then used to evaluate the GLUE dataset in <xref ref-type="sec" rid="s4-4-3">Section 4.4.3</xref>. MLPerf inference edge uses BERT-large, hence, it is used when comparing accelerators in <xref ref-type="sec" rid="s4-5">Section 4.5</xref>.</p>
<p>We simulated various sequence lengths by truncating a large passages of text to a specified sequence length to obtain the delay and energy metrics in <xref ref-type="sec" rid="s4-3">Sections 4.3</xref>, <xref ref-type="sec" rid="s4-4-1">4.4.1</xref>, <xref ref-type="sec" rid="s4-4-2">4.4.2</xref>.</p>
<p>
<xref ref-type="sec" rid="s4-4-3">Section 4.4.3</xref> uses the General Language Understanding Evaluation (GLUE) benchmark, which is a collection of NLP datasets for various NLP tasks. The GLUE dataset is composed of the Stanford Sentiment Treebank (SST-2), the Microsoft Research paraphrase Corpus (MRPC), the Quora Question Pairs (QQP) dataset, the Semantic Textual Similarity benchmark (STSB), Multi-Genre Natural Language Inference (MNLI) Corpus, Question Natural Language Inference (QNLI) dataset, Recognizing Textual Entailment (RTE), Winograd Natural Language Inference (WNLI) Schema Challenge.</p>
<p>To evaluate and compare with other hardware implementations, <xref ref-type="sec" rid="s4-5">Section 4.5</xref> uses the setup of MLPerf, an industry-standard machine learning benchmark to evaluate hardware devices in a myriad of machine learning tasks. MLPerf Inference Edge, in particular, specifically evaluates machine learning systems in edge devices during inference. One of the categories in MLPerf Inference Edge is the language processing task that uses BERT-large and the Stanford Question Answering Dataset (SQuAD) 1.1. The QNLI dataset of the GLUE benchmark is derived from the SQuAD dataset. The SQuAD 1.1 dataset is a reading comprehension dataset consisting of 100k &#x2b; question and answer pairs where the answers to the questions can be obtained from a set of 500 &#x2b; Wikipedia articles.</p>
</sec>
</sec>
<sec id="s4-3">
<title>4.3 Multi-Head Attention Evaluation</title>
<p>This section presents the multi-head attention evaluation of iMTransformer compared to the GPU baseline. The input sequence length affects the computational demands of transformers. By using crossbars and attention caches, iMTransformer can improve the execution time of transformers, particularly the MHA, as the sequence length increases (<xref ref-type="sec" rid="s4-3-1">Section 4.3.1</xref>). Furthermore, bidirectional MHA can be parallelized to achieve higher speedups (<xref ref-type="sec" rid="s4-3-2">Section 4.3.2</xref>). Energy consumption can be further improved by using locality-based sparsity (<xref ref-type="sec" rid="s4-3-3">Section 4.3.3</xref>) and content-based sparsity (<xref ref-type="sec" rid="s4-3-4">Section 4.3.4</xref>). We assume a CMOS-based crossbar in this evaluation.</p>
<sec id="s4-3-1">
<title>4.3.1 Effect of Increasing Sequence Length</title>
<p>We first evaluate the unparalleled MHA with respect to the GPU baseline. The unparallelized MHA implementation reduces the memory transfer overhead <italic>via</italic> PIM, reduces the number of computations by caching the keys and values, and increases parallelism by using crossbars. <xref ref-type="fig" rid="F6">Figures 6A&#x2013;C</xref> shows the improvement of implementing MHA in iMTransformer when running a bidirectional transformer for inference. At a sequence length of <italic>n</italic> &#x3d; 64, the standard speedup (shown as red bars in <xref ref-type="fig" rid="F6">Figure 6A</xref>) of running the transformer using crossbars is 10.6&#xd7;. The GPU fully utilizes the memory and compute units at this sequence length and achieves optimal performance.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Comparison between CMOS-based iMTransformer and GPU Baseline in terms of the speedup <bold>(A)</bold>, energy <bold>(B)</bold> and EDP <bold>(C)</bold> evaluation results for multi-head attention (with and without parallelism) for varying sequence length. Energy improvements of MHA without parallelism for locality-based <bold>(D)</bold> and content-based <bold>(E)</bold> sparsity.</p>
</caption>
<graphic xlink:href="felec-03-847069-g006.tif"/>
</fig>
<p>As the sequence length <italic>n</italic> increases, the GPU&#x2019;s memory bandwidth and the compute limitations are reached. At <italic>n</italic> &#x3d; 256, the execution time of the baseline starts to grow quadratically by approximately <italic>n</italic>
<sup>2</sup>/256. By caching the keys and values and using crossbars, iMTransformer only grows linearly. Caching the keys and values reduces the required number of matrix-vector multiplications for <bold>W</bold>
<sup>
<bold>Q</bold>
</sup>, <bold>W</bold>
<sup>
<bold>K</bold>
</sup>, and <bold>W</bold>
<sup>
<bold>V</bold>
</sup> by <italic>n</italic>, thus reducing the delay by <italic>n</italic>. Since the PUs contains static weights, their execution time still grows by <italic>n</italic>, as we need to compute the projection of the <italic>q</italic>&#x2032;, <bold>K</bold>&#x2032;, and <bold>V</bold>&#x2032; for each segment. The number of required columns increases for <italic>K</italic> while the number of required rows increases for <bold>V</bold> as <italic>n</italic> increases. Since the columns in the crossbars can be treated independently, &#x2308;<italic>n</italic>/64&#x2309; 64 &#xd7; 64 arrays can be used to compute the matrix-vector multiplications for <italic>K</italic> in parallel. Increasing the number of rows in <bold>V</bold> requires adding partial sums from the crossbars, incurring additional latency. This additional latency, however, is still not significant at <italic>n</italic> &#x3d; 4096. Hence, the complexity of iMTransformer is <italic>O</italic>(<italic>n</italic>). Thus, iMTransformer achieves a speedup close to <italic>n</italic>/256 compared to a GPU-based solution as shown by the red bars in <xref ref-type="fig" rid="F6">Figure 6A</xref>.</p>
</sec>
<sec id="s4-3-2">
<title>4.3.2 Improvement due to Model Parallelism</title>
<p>For the encoder layers, latency can be improved by introducing more parallelism <italic>via</italic> duplicating the crossbars in each head by <italic>p</italic> (<xref ref-type="sec" rid="s3-2-2">Section 3.2.2</xref>). The projection operation is parallelized but is bottlenecked by the write operation. SDPA, on the other hand, results in <italic>p</italic>&#xd7; additional improvement in the execution time. By using crossbars and parallelization (<italic>p</italic> &#x3d; <italic>n</italic>), we can reduce the time complexity of the projection to <italic>O</italic>(<italic>n</italic>) and SDPA to <italic>O</italic> (1). Accounting for the projections and SDPA, the latency improvement is shown as yellow bars in <xref ref-type="fig" rid="F6">Figure 6A</xref>.</p>
<p>Memory and power, however, are bounded. As an example, we set the memory size of each attention head to 15MB, shown as blue bars in <xref ref-type="fig" rid="F6">Figure 6A</xref> as bidirectional MHA BM (Bounded Memory). Once the memory limit of iMTransformer is reached, the execution time returns to <italic>O</italic> (<italic>n</italic>/<italic>p</italic>), where <italic>p</italic> is the number of AH Mats used to represent a single attention head to achieve attention-level parallelism through duplication. At <italic>n</italic> &#x3d; 64, this speedup equates to 682&#xd7; speedup (in which <inline-formula id="inf8">
<mml:math id="m9">
<mml:mo>&#x2248;</mml:mo>
<mml:mn>10</mml:mn>
<mml:mo>&#xd7;</mml:mo>
</mml:math>
</inline-formula> speedup is from computing in-memory in crossbars, and around <inline-formula id="inf9">
<mml:math id="m10">
<mml:mo>&#x2248;</mml:mo>
<mml:mn>64</mml:mn>
<mml:mo>&#xd7;</mml:mo>
</mml:math>
</inline-formula> speedup from parallelization). At <italic>n</italic> &#x3d; 4096, a speedup of five orders of magnitude is achieved (&#xd7;10 from the standard implementation, 100&#xd7; from duplicating the crossbars, and 100&#xd7; from attention duplication).</p>
<p>However, duplicating crossbars increases the number of writes and requires communication between AH Mats, resulting in lower energy improvements. In the parallel scenario (<xref ref-type="fig" rid="F6">Figure 6B</xref>), increasing attention-level parallelism with respect to sequence length <italic>n</italic> also requires <italic>n</italic>&#xd7; more writes (because <italic>p</italic> &#x3d; <italic>n</italic>), resulting in a significant degradation in energy improvement. If <italic>n</italic> &#x3c; <italic>p</italic>, only <italic>n</italic>&#xd7; more writes are required. If <italic>n</italic> &#x2265; <italic>p</italic>, there are <italic>p</italic>&#xd7; more writes. The memory requirement also increases by a factor of <italic>n</italic>.</p>
<p>Having more parallelism results in a higher speedup but lower energy gains. However, per <xref ref-type="fig" rid="F6">Figure 6C</xref>, increased parallelism translates to better EDP owing to the exponential decrease of delay improvement despite the linear increase in energy improvement.</p>
</sec>
<sec id="s4-3-3">
<title>4.3.3 Improvement due to In-Memory Locality-Based Sparse Attention</title>
<p>Different transformer models have used different attention patterns to reduce the space and computational complexity of the transformer networks. In iMTransformer, the computation of attention scores is highly parallelized. Therefore changing the attention pattern does not reduce latency. However, because of fewer parallel computations, energy consumption is reduced. <xref ref-type="fig" rid="F6">Figure 6D</xref> shows the energy improvement of using full, masked, strided, sliding window, dilated, and sliding window &#x2b; strided attention patterns as compared to the full attention pattern implementation in the GPU.</p>
<p>Compared to the full MHA (red bars in <xref ref-type="fig" rid="F6">Figure 6D</xref>), the masked MHA (orange bars in <xref ref-type="fig" rid="F6">Figure 6D</xref>) reduces the number of rows in CU-V that needs to be queried by 2&#xd7; regardless of the sequence length. This is equivalent to turning off the ADCs in CU-K. Masked MHA achieves a speedup of 1.9&#xd7; and 2&#xd7; compared to the full SDPA at a sequence length 512 and 4096, respectively. The energy consumption of the strided window MHA (shown as yellow bars in <xref ref-type="fig" rid="F6">Figure 6D</xref>) depends on the stride length. (In our example, it is 4). Hence, it can achieve up to 4&#xd7; energy improvement than the full MHA. For sequence lengths of 512 and 4096, sliding window attention (shown as green bars in <xref ref-type="fig" rid="F6">Figure 6D</xref>) achieves energy improvements of 3.43&#xd7; and 3.91&#xd7;, respectively. The sliding window MHA achieves higher energy improvement as the sequence length increases.</p>
<p>The sliding window MHA only focuses on <italic>w</italic> sequence elements for each iteration, where <italic>w</italic> is the sliding window length. Hence, the sliding window attention increases linearly as the sequence length increases. Compared to the full MHA, sliding window MHA results in an energy improvement of 13.92&#xd7; at a sequence length equal to 512, and energy improvement of 106&#xd7; at a sequence length equal to 4096. As some sequence elements are skipped, the dilated sliding window has a small improvement against the sliding window MHA and achieves energy improvements of 15.51&#xd7; and 118.14&#xd7; for sequence lengths of 512 and 4096, respectively. Strided MHA consumes more energy than the sliding window MHA and dominates the sliding window &#x2b; strided MHA. The dilated sliding window attention pattern shows the highest energy improvement, followed by the sliding window attention pattern among the six different patterns shown. However, as stated before, which attention pattern should be used is application-dependent.</p>
</sec>
<sec id="s4-3-4">
<title>4.3.4 Improvement due to In-Memory Content-Based Sparsity</title>
<p>The accuracy of using LSH followed by SDP (LSH &#x2b; SDP) is dependent on signature length. We have found that comparable accuracies with SDP can be achieved when using a signature length of 1024 bits with a <italic>k</italic>-nearest neighbor of <italic>k</italic> &#x3d; 16 on the General Language Understanding Evaluation (GLUE) dataset (<xref ref-type="bibr" rid="B64">Wang et al., 2018</xref>). This setup is also similar to using 64 buckets of four hashes each or 1024 &#x3d; 64 &#x22c5; 2<sup>4</sup> bits for the angular LSH used in the Reformer network (<xref ref-type="bibr" rid="B27">Kitaev et al., 2020</xref>). A pre-trained BERT (<xref ref-type="bibr" rid="B12">Devlin et al., 2018</xref>) is used for evaluation. The GLUE dataset is a common benchmark used in transformers (<xref ref-type="bibr" rid="B12">Devlin et al., 2018</xref>; <xref ref-type="bibr" rid="B32">Lan et al., 2019</xref>).</p>
<p>Since we are only appending the LSH in a pre-trained network, an 11% (1.11&#xd7;) increase in delay is incurred because of the CAM search. LSH reduces the number of dot-products required as the sequence length increases. Latency overhead still outweighs latency improvement due to fewer attention computations that must be performed at a sequence length of <italic>n</italic> &#x3d; 4096. However, as CAM searches require fewer numbers of bits and are more energy-efficient than searching <italic>via</italic> dot-product on crossbars, this results in an exponential improvement in energy as the sequence length increases (<xref ref-type="fig" rid="F6">Figure 6E</xref>). The bars below the dashed line (speedup &#x3d; 1) have worse execution times, representing a slowdown. LSH is only advisable for longer sequence lengths (greater than 256).</p>
</sec>
</sec>
<sec id="s4-4">
<title>4.4 End-to-End Evaluation</title>
<p>We evaluate the end-to-end accuracy, delay, and energy of iMTransformer in implementing the transformer network. <xref ref-type="sec" rid="s4-4-1">Section 4.4.1</xref> considers the end-to-end energy and delay evaluation of iMTransformer. <xref ref-type="sec" rid="s4-4-2">Section 4.4.2</xref> discusses the improvement in energy consumption by iMTransformer using FeFET-based crossbars for high-read operations, and CMOS-based crossbars for high-write operations. Finally, <xref ref-type="sec" rid="s4-4-3">Section 4.4.3</xref> reports the effect of device-to-device resistance variation on application-level accuracy.</p>
<sec id="s4-4-1">
<title>4.4.1 Energy and Delay Evaluation</title>
<p>
<xref ref-type="fig" rid="F7">Figures 7A,B</xref> shows the delay and energy improvement of feedforward and MHA with parallelism and LSH enhancements on the Vanilla and BERT-based transformer at sequence lengths <italic>n</italic> &#x3d; 512 and <italic>n</italic> &#x3d; 4096. The standard implementation (without attention-level parallelism) achieves a speedup of 16&#xd7; and 6.4&#xd7; for the vanilla transformer and BERT, respectively, when <italic>n</italic> &#x3d; 512. This increases to 305.6&#xd7; and 167.2&#xd7; at <italic>n</italic> &#x3d; 4096. For the vanilla transformer, only the encoder layers can be parallelized, as opposed to bidirectional transformers such as BERT, where all layers can be parallelized. Therefore, as the sequence length increases, the speedup gained from attention parallelization is smaller for vanilla transformers than bidirectional transformers. LSH slows down the iMTransformer as shown in <xref ref-type="fig" rid="F7">Figure 7A</xref> because of the additional hashing operation achieving a 456&#xd7; speedup for Vanilla transformer and 39.49K&#xd7; speedup for BERT for <italic>n</italic> &#x3d; 4096.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Latency <bold>(A)</bold> and energy <bold>(B)</bold> improvements of evaluating the MHA with and without model parallelism (MP) and sparsity and Feedforward units of Vanilla transformers and BERT with sequence length of 512 and 4096 using CMOS-based iMTransformer compared to GPU Baseline. End-to-end improvement <bold>(C)</bold> of CMOS-based iMTransformer for Vanilla transformers and BERT with sequence length of 512 and 4096 compared to GPU baseline <bold>(D)</bold> End-to-end improvement of iMTransformer for Vanilla transformers with sequence length of 512 using CMOS, FeFET and CMOS &#x2b; FeFET iMTransformer implementation compared to the GPU baseline. <bold>(E)</bold> Effects of crossbar resistance variation on the accuracy the GLUE dataset.</p>
</caption>
<graphic xlink:href="felec-03-847069-g007.tif"/>
</fig>
<p>The energy improvement of the standard implementation of iMTransformer for the vanilla transformer is 16.84&#xd7; at <italic>n</italic> &#x3d; 512. This increases to 56.78&#xd7; at <italic>n</italic> &#x3d; 4096 because of fewer number of computations due to caching compared to the GPU implementation. For the standard implementation of BERT, the iMTransformer has an energy improvement of 6.78&#xd7; at <italic>n</italic> &#x3d; 512 which increases to 31.16 &#xd7; <italic>n</italic> &#x3d; 4096. The parallel implementation is faster than the standard implementation but requires hardware duplication, thus, consuming more energy. LSH improves the energy consumption to 40.18&#xd7; at <italic>n</italic> &#x3d; 4096 compared to the GPU implementation.</p>
<p>By including layer normalization as shown in <xref ref-type="fig" rid="F7">Figure 7C</xref>, iMTransformer can achieve a delay improvement of 9.01&#xd7; for Vanilla Transformer at <italic>n</italic> &#x3d; 512, 13.71&#xd7; for BERT at <italic>n</italic> &#x3d; 512, 76&#xd7; for Vanilla Transformer at <italic>n</italic> &#x3d; 4096 and 199&#xd7; for BERT at <italic>n</italic> &#x3d; 4096. iMTransformer also achieves an energy improvement of 5.83&#xd7; for Vanilla Transformer at <italic>n</italic> &#x3d; 512, 1.49&#xd7; for BERT at <italic>n</italic> &#x3d; 512, 60.4&#xd7; for Vanilla Transformer at <italic>n</italic> &#x3d; 4096 and 33.6&#xd7; for BERT at <italic>n</italic> &#x3d; 4096. Algorithmic replacements for layer normalization that are more hardware friendly are needed to improve the iMTransformer accelerator further.</p>
</sec>
<sec id="s4-4-2">
<title>4.4.2 Improvement due to Using Emerging Technology</title>
<p>Based on the data in <xref ref-type="table" rid="T2">Table 2</xref>, CMOS-based crossbars are 1.02&#xd7; faster than FeFET-based crossbar. On the other hand, FeFET-based crossbars have 4.12&#xd7; lower read energy than CMOS-based crossbars. CMOS-based crossbars have faster writes and lower write energy than FeFET-based crossbars. As discussed before, CMOS crossbars also have higher endurance than FeFET-based crossbars. Because of this, we utilize CMOS-based crossbars for crossbars that require a high number of write operations and FeFET crossbars for crossbars that do not require write operations. In the CMOS-FeFET hybrid iMTransformer, CMOS is used for attentional caches, while FeFETs are used to store trained weights.</p>
<p>
<xref ref-type="fig" rid="F7">Figure 7D</xref> shows the end-to-end latency and energy improvements of iMTransformer using CMOS, FeFET and CMOS &#x2b; FeFET hybrid iMTransformer implementations when implementing the Vanilla Transformer at <italic>n</italic> &#x3d; 512. The CMOS-based iMTransformer achieves an end-to-end speedup and energy improvement of 9.01&#xd7; and 5.83&#xd7;, respectively. In contrast, the FeFET-based iMtransformer achieves 4.68&#xd7; end-to-end latency and 12.03&#xd7; end-to-end energy improvements. The CMOS-based iMTransformer performs 1.93&#xd7; faster than the FeFET-based iMTransformer. However, the FeFET based iMTransformer has 2.06&#xd7; lower energy consumption than the CMOS-based transformer.</p>
<p>For the CMOS &#x2b; FeFET hybrid iMTransformer implementation, we can achieve an 8.96&#xd7; speedup, which is close to the latency improvement of the CMOS iMTransformer implementation (9.01&#xd7;). The CMOS &#x2b; FeFET hybrid iMTransformer implementation achieves a 12.57&#xd7; energy improvement, higher than either the CMOS or the FeFET iMTransformer implementation. This is because we utilize CMOS crossbars (which are more energy efficient for write operations) for attentional caches and FeFET crossbars (which are more energy efficient for read operations) for storing trained weights. Thus the CMOS &#x2b; FeFET iMTransformer implementation exhibits a much better energy-delay-product than using a CMOS or FeFET iMTransformer implementation alone.</p>
</sec>
<sec id="s4-4-3">
<title>4.4.3 Accuracy Evaluation</title>
<p>Device programming methods for writing specific values to FeFETs can result in different stochastic variations of the stored resistance in crossbars. We evaluate the effect of the resistance variation in the crossbar array on the accuracy of the GLUE dataset using the BERT-base model. Crossbar resistance variation is usually modelled as a log-normal distribution (<xref ref-type="bibr" rid="B37">Li P. et al., 2021</xref>; <xref ref-type="bibr" rid="B33">Lastras-Monta&#xf1;o et al., 2021</xref>) <italic>R</italic>
<sub>
<italic>d2d</italic>
</sub> &#x3d; <italic>R</italic>
<sub>
<italic>ideal</italic>
</sub>
<italic>e</italic>
<sup>
<italic>&#x3b8;</italic>
</sup> where <inline-formula id="inf10">
<mml:math id="m11">
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. The resistance variation of crossbars can affect the accuracy of implementing the transformer network in iMTransformer. <xref ref-type="fig" rid="F7">Figure 7E</xref> shows the effect of the resistance variation on the GLUE benchmark. At <italic>&#x3c3;</italic> &#x3c; 0.3 (equivalent to 12.7<italic>%</italic> either below or above the mean), we achieve iso-accuracy for almost all the datasets in the GLUE benchmark. The accuracy then continues to drop until <italic>&#x3c3;</italic> &#x3e; 0.5 or at 19.15<italic>%</italic> either below or above the mean. This means that the maximum allowable resistance variation in the crossbar must be below &#xb1;12.7<italic>%</italic>.</p>
</sec>
</sec>
<sec id="s4-5">
<title>4.5 Comparison With Other Accelerators</title>
<p>To provide a comparison with GPU baselines and PIM-based accelerators for transformer networks, we use the MLPerf Inference Single Stream setup. The MLPerf uses BERT-Large and the SQuAD 1.1 dataset. The SQuAD 1.1 dataset has a maximum sequence length of 384, with most of the questions/answers in the dataset in the 100&#x2013;200 sequence length range. <xref ref-type="table" rid="T3">Table 3</xref> shows the comparison between results from the best MLPerf baseline, the GPU baseline ReTransformer, and the CMOS-based and CMOS-FeFET hybrid Transformer baselines. The MLPerf Inference Edge results represent the best <italic>Single Stream</italic> (batch size of one) results<xref ref-type="fn" rid="fn1">
<sup>1</sup>
</xref> as of the time of this writing. It uses Intel&#xae; Xeon&#xae; Platinum 8358, NVIDIA A100-SXM-80GB using TensorRT 8.0.1 and CUDA 11.3. The Titan X GPU (the GPU used in this paper as the main baseline) throughput is obtained by averaging all dataset samples. The ReTransformer results are obtained from the best latency performance (81.85 GOps/<italic>s</italic>), and energy efficiency (467.68/<italic>s</italic>/<italic>W</italic>) reported in <xref ref-type="bibr" rid="B67">Yang X. et al. (2020)</xref>. The iMTransformer results are obtained by the hardware implementation for BERT-Large model parallelization and attention caching. iMTransformer did not use sparsity because the SQuAD 1.1 dataset has short sentences (low sequence length); hence, sparsity will not improve the results significantly.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Comparison of the leading MLPerf Inference - Edge results, the Titan X baseline, ReTransformer and the CMOS-based and Hybrid iMTransformer using MLPerf setting. MLPerf uses BERT-large and the SQuAD 1.1 dataset with a maximum sequence length of 384.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Accelerator</th>
<th align="center">Throughput (Samples/<italic>s</italic>)</th>
<th align="center">Throughput/W (Samples/<italic>s</italic>/<italic>W</italic>)</th>
<th align="center">Throughput/J (Samples/<italic>s</italic>/<italic>J</italic>)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">MLPerf (Best delay)</td>
<td align="char" char=".">649.35</td>
<td align="center">Not available</td>
<td align="center">Not available</td>
</tr>
<tr>
<td align="left">Titan X (Baseline)</td>
<td align="char" char=".">204.84</td>
<td align="center">15.76</td>
<td align="char" char=".">3.23&#xa0;K</td>
</tr>
<tr>
<td align="left">ReTransformer</td>
<td align="char" char=".">2.73</td>
<td align="center">15.60</td>
<td align="center">42.59</td>
</tr>
<tr>
<td align="left">iMTransformer-CMOS</td>
<td align="char" char=".">2.25 K</td>
<td align="center">23.48</td>
<td align="char" char=".">52.83&#xa0;K</td>
</tr>
<tr>
<td align="left">iMTransformer-Hybrid</td>
<td align="char" char=".">2.23 K</td>
<td align="center">124.8</td>
<td align="char" char=".">278&#xa0;K</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>As shown in <xref ref-type="table" rid="T3">Table 3</xref>, the MLPerf Inference Edge achieves an average throughput of 649.35 samples/sec. The Titan X GPU baseline has a slower GPU and slower memory bandwidth leading to a 204.84 samples/sec throughput. The ReTransformer also has smaller operations/second than MLPerf GPU (A100) and Titan X GPU hence the smaller throughput of 2.73 samples/sec. However, this implementation for ReTransformer is not optimized for BERT and does not utilize the available parallelism of Transformers. The iMTransformer uses model parallelization and attention caching which are not present in the ReTransformer. The iMTransformer achieves a throughput of 2.25 K for the CMOS iMTransformer implementation and 2.23 K for the CMOS-FeFET hybrid iMTransformer implementation, which are about 11&#xd7; better than Titan X and 3.43&#xd7; improvement over the current MLPerf state of the art results.</p>
<p>The energy results are summarized in <xref ref-type="table" rid="T3">Table 3</xref>. The energy results are not available for the MLPerf Inference Edge baseline and hence are not reported in <xref ref-type="table" rid="T3">Table 3</xref>. The Titan X does not achieve its peak compute utilization and uses an average dynamic power of 13&#xa0;W (peaking at 35&#xa0;W) to run the transformer network, giving a throughput/W of 15.76 samples/s/W and throughput/J of 3.23 K samples/s/J as shown in <xref ref-type="table" rid="T3">Table 3</xref> iMTransformer reduces the memory transfer overhead by using in-memory computing compared to Titan X and reduces the computation <italic>via</italic> attention caching compared to ReTransformer. iMTransformer achieved around 1.5&#xd7; energy improvement compared to Titan X and ReTransformer. The CMOS-FeFET hybrid iMTransformer achieves an additional 5.31&#xd7; energy improvement leading to 125 samples/s/W. Given in-memory computing, model parallelization, attention caching, and using a CMOS-FeFET hybrid iMTransformer implementation, we achieve 278K throughput/J, which is 5.26&#xd7; better than the CMOS iMTransformer and 86&#xd7; better than the Titan X (the baseline GPU).</p>
</sec>
</sec>
<sec id="s5">
<title>5 Discussion</title>
<p>In this work, we introduced iMTransformer, an in-memory computing-based transformer network accelerator (iMTransformer) that uses FeFET-based and CMOS-based crossbars and CAMs to accelerate transformer networks, specifically the multi-head attention computations.</p>
<p>iMTransformer achieves latency and energy improvements by (1) reducing the memory bottleneck <italic>via</italic> computing-in-memory, (2) reducing the number of computations by storing reusable data in crossbars, and (3) maximizing the parallelism for bidirectional MHA and encoder-decoder MHA. Computing-in-memory alleviates the memory transfer bottleneck by reducing the need to move data from the memory to the compute unit and vice versa. By using crossbars as attentional caches, we can keep data computed from previous time steps to be reused for succeeding time steps. Finally, different types of MHA have parallelism characteristics that can be exploited, such as the bi-directionality of encoder MHAs and parallelizing the computation of encoder-decoder keys and values across different layers.</p>
<p>Furthermore, iMTransformer improves energy efficiency by (1) using an attention selector, (2) exploiting content-based sparsity using CAMs, and (3) using CMOS and FeFET devices. The attention selector allows masking and locality-based sparse attention. The attention selector reduces the number of computations and activation of unnecessary ADCs based on the attention patterns. CAMs are employed to implement LSH for realizing content-based sparsity. Though the LSH computations incur latency overhead, they significantly reduce energy consumption as the sequence length increases. Finally, non-volatile FeFET-based crossbars are used for crossbars that involve highly frequent read operations in the feedforward tile and processing unit. Because of their high write requirements, CMOS-based crossbars are used for the attentional caching units.</p>
<p>For the Vanilla Transformer at sequence length of 512, <xref ref-type="fig" rid="F8">Figure 8A</xref> shows a delay improvement of 7.7&#xd7; for the standard implementation compared to the GPU baseline. Including model parallelization further improves the end-to-end delay by 1.17&#xd7; or an end-to-end improvement of 9.01&#xd7; compared to the GPU baseline. Introducing sparsity does not improve the delay while using the CMOS-FeFET hybrid iMTransformer reduces the end-to-end improvement to 8.96&#xd7;. The end-to-end energy improvement of the standard implementation is 7.81&#xd7; compared to the GPU implementation. Because of the increase in the write operations, the energy is reduced to 5.71&#xd7; and slightly improves with sparsity at 5.83&#xd7; compared to the GPU baseline. The standard implementation achieves a 60.17&#xd7; EDP improvement while adding all the enhancements improves the EDP by 112.66&#xd7; compared to the GPU.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Delay, Energy and EDP improvement for the Vanilla Transformer <bold>(A)</bold> and BERT <bold>(B)</bold> with sequence length of 512 with respect to additional enhancement: Model Parallelism (MP), Locality Sensitive Hashing (LSH) and using FeFET devices.</p>
</caption>
<graphic xlink:href="felec-03-847069-g008.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="F8">Figure 8B</xref> shows the delay, energy, and EDP improvement of BERT for a sequence length of 512. The standard implementation of iMTransformer achieves a 4.68&#xd7; energy improvement compared to the GPU implementation. Because of its bi-directional nature, BERT greatly benefits from model parallelism and achieves 13.76&#xd7; energy improvement (a 2.94&#xd7; increase) compared to the GPU baseline. Introducing sparsity and using the hybrid iMTransformer slightly reduces the improvement to 13.71&#xd7;. The standard implementation has an energy improvement of 4.78&#xd7;. Because of the additional energy due to writes and greater utilization of model parallelism, the implementation with model parallelism achieves a 1.78&#xd7; improvement while adding sparsity further reduces the implementation to 1.49&#xd7;. However, introducing the use of hybrid transformer improves the energy to 8.95&#xd7; when compared to the GPU. The standard implementation achieves a 22.34&#xd7; EDP improvement while adding all the enhancements improves the EDP to 122.65&#xd7; compared to the GPU baseline.</p>
<p>For both Vanilla and BERT, introducing model parallelization improves the delay. However, because BERT is an encoder-type transformer, it can benefit from model parallelization more than the Vanilla transformer. However, model parallelization comes at the cost of more energy usage. Using FeFET devices for static weights slightly slows down the iMTransformer but greatly improves the energy. The results in <xref ref-type="fig" rid="F8">Figure 8</xref> are only for a sequence length of 512. Greater improvements can be achieved at longer sequence lengths. Using the MLPerf benchmark, the hybrid iMTransformer can query 2.23 k samples/sec at 125 samples/s/W as shown in <xref ref-type="table" rid="T3">Table 3</xref>.</p>
<p>In this paper, we have introduced iMTransformer, an in-Memory computing Transformer Network accelerator. We have been able to greatly improve the delay and energy consumption of the transformer&#x2019;s attention mechanism and feedforward layers. Transformer networks are evolving, and different variants for various applications have been developed in the past few years. Transformer network accelerators must adapt to different types of attention mechanisms and transformer network configuration. This work is catered to the original transformer network, where the input is a sequence. This makes transformer networks ideal for NLP applications. Because of the transformer&#x2019;s ability to model long-range dependencies, the sequence length can continue to increase and require more sophisticated algorithms and accelerators.</p>
</sec>
</body>
<back>
<sec id="s6">
<title>Data Availability Statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://gluebenchmark.com/">https://gluebenchmark.com/</ext-link>, <ext-link ext-link-type="uri" xlink:href="https://rajpurkar.github.io/SQuAD-explorer/">https://rajpurkar.github.io/SQuAD-explorer/</ext-link>.</p>
</sec>
<sec id="s7">
<title>Author Contributions</title>
<p>AL performed the application level and architectural simulations and was the primary proponent of the conception of the idea. MS performed the simulations for the peripherals and interconnects. AK performed the crossbar simulations. XY performed the CAM simulations. MN and XH supervised all experiments. AL wrote the manuscript with input from all authors. All authors reviewed the document.</p>
</sec>
<sec id="s8">
<title>Funding</title>
<p>This work was supported in part by ASCENT, one of six centers in JUMP, a Semiconductor Research Corporation (SRC) program sponsored by DARPA. The funder was not involved in the study, design, collection, analysis, interpretation of data, the writing of this article or the decision to submit it for publication.</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn id="fn1">
<label>1</label>
<p>Results taken 15 February 2022</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Beltagy</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Peters</surname>
<given-names>M. E.</given-names>
</name>
<name>
<surname>Cohan</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <source>Longformer: The Long-Document Transformer</source>. </citation>
</ref>
<ref id="B2">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Beyer</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dunkel</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Trentzsch</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Muller</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hellmich</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Utess</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <source>FeFET: A Versatile CMOS Compatible Device with Game-Changing Potential</source>. </citation>
</ref>
<ref id="B3">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Boes</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Van hamme</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Audiovisual Transformer Architectures for Large-Scale Classification and Synchronization of Weakly Labeled Audio Events</article-title>,&#x201d; in <source>Proceedings of the 27th ACM International Conference on Multimedia</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Amsaleg</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Huet</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Larson</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Gravier</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Hung</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ngo</surname>
<given-names>C.-W.</given-names>
</name>
<etal/>
</person-group> (<publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>1961</fpage>&#x2013;<lpage>1969</lpage>. <pub-id pub-id-type="doi">10.1145/3343031.3350873</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Brown</surname>
<given-names>T. B.</given-names>
</name>
<name>
<surname>Mann</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ryder</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Subbiah</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kaplan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Dhariwal</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <source>Language Models Are Few-Shot Learners</source>. </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Challapalle</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Rampalli</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jao</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Ramanathan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sampson</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Narayanan</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>FARM: A Flexible Accelerator for Recurrent and Memory Augmented Neural Networks</article-title>. <source>J. Sign Process. Syst.</source> <volume>92</volume>, <fpage>1247</fpage>&#x2013;<lpage>1261</lpage>. <pub-id pub-id-type="doi">10.1007/s11265-020-01555-w</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>P.-Y.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018a</year>). <article-title>NeuroSim: A Circuit-Level Macro Model for Benchmarking Neuro-Inspired Architectures in Online Learning</article-title>. <source>IEEE Trans. Comput.-Aided Des. Integr. Circuits Syst.</source> <volume>37</volume>, <fpage>3067</fpage>&#x2013;<lpage>3080</lpage>. <pub-id pub-id-type="doi">10.1109/tcad.2018.2789723</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yin</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Niemier</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X. S.</given-names>
</name>
</person-group> (<year>2018b</year>). &#x201c;<article-title>Design and Optimization of FeFET-Based Crossbars for Binary Convolution Neural Networks</article-title>,&#x201d; in <source>2018 Design, Automation Test in Europe Conference Exhibition (DATE)</source> (<publisher-name>IEEE</publisher-name>), <fpage>1205</fpage>&#x2013;<lpage>1210</lpage>. <pub-id pub-id-type="doi">10.23919/date.2018.8342199</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Child</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gray</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Radford</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sutskever</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2019</year>). <source>Generating Long Sequences with Sparse Transformers</source>. </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chua-Chin Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chia-Hao Hsu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chi-Chun Huang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Jun-Han Wu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>A Self-Disabled Sensing Technique for Content-Addressable Memories</article-title>. <source>IEEE Trans. Circuits Syst.</source> <volume>57</volume>, <fpage>31</fpage>&#x2013;<lpage>35</lpage>. <pub-id pub-id-type="doi">10.1109/tcsii.2009.2037995</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chung</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kwon</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Jeon</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). &#x201c;<article-title>Extremely Low Bit Transformer Quantization for On-Device Neural Machine Translation</article-title>,&#x201d; in <source>Findings of the Association for Computational Linguistics: EMNLP 2020</source> (<publisher-name>Online: Association for Computational Linguistics</publisher-name>), <fpage>4812</fpage>&#x2013;<lpage>4826</lpage>. <pub-id pub-id-type="doi">10.18653/v1/2020.findings-emnlp.433</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dai</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Carbonell</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>Q. V.</given-names>
</name>
<name>
<surname>Salakhutdinov</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Transformer-XL: Attentive Language Models beyond a Fixed-Length Context</article-title>. </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Devlin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>M.-W.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Toutanova</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding</article-title> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dosovitskiy</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Beyer</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Kolesnikov</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Weissenborn</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhai</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Unterthiner</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>An Image Is worth 16x16 Words: Transformers for Image Recognition at Scale</article-title> </citation>
</ref>
<ref id="B14">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Fedus</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zoph</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Shazeer</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2021</year>). <source>Switch Transformers: Scaling to Trillion Parameter Models with Simple and Efficient Sparsity</source>. </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gokmen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Vlasov</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Acceleration of Deep Neural Network Training with Resistive Cross-Point Devices: Design Considerations</article-title>. <source>Front. Neurosci.</source> <volume>10</volume>, <fpage>333</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2016.00333</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Huangfu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>RADAR: A 3D-ReRAM Based DNA Alignment Accelerator Architecture</article-title>,&#x201d; in <conf-name>Proceedings of the 55th Annual Design Automation Conference</conf-name> (<publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>6</lpage>. <comment>Article 59 in DAC &#x2019;18</comment>. <pub-id pub-id-type="doi">10.1109/dac.2018.8465882</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jeloka</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Akesh</surname>
<given-names>N. B.</given-names>
</name>
<name>
<surname>Sylvester</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Blaauw</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>A 28 Nm Configurable Memory (TCAM/BCAM/SRAM) Using Push-Rule 6T Bit Cell Enabling Logic-In-Memory</article-title>. <source>IEEE J. Solid-state Circuits</source> <volume>51</volume>, <fpage>1009</fpage>&#x2013;<lpage>1021</lpage>. </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jerry</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Dutta</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kazemi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ni</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>P.-Y.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>A Ferroelectric Field Effect Transistor Based Synaptic Weight Cell</article-title>. <source>J. Phys. D Appl. Phys.</source> <volume>51</volume>, <fpage>434001</fpage>. </citation>
</ref>
<ref id="B19">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Junczys-Dowmunt</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Grundkiewicz</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Dwojak</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hoang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Heafield</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Neckermann</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <source>Marian: Fast Neural Machine Translation in C&#x2b;&#x2b;</source>. </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kaiser</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Nachum</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Roy</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Learning to Remember Rare Events</article-title>,&#x201d; in <conf-name>5th International Conference on Learning Representations</conf-name>. </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>In-Memory Processing Paradigm for Bitwise Logic Operations in STT-MRAM</article-title>. <source>IEEE Trans. Magn.</source> <volume>53</volume>, <fpage>1</fpage>&#x2013;<lpage>4</lpage>. <pub-id pub-id-type="doi">10.1109/tmag.2017.2703863</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kaplan</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Yavits</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ginosar</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2018</year>). <source>RASSA: Resistive Pre-alignment Accelerator for Approximate DNA Long Read Mapping</source>, <fpage>44</fpage>&#x2013;<lpage>54</lpage>. </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karam</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Puri</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ghosh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bhunia</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Emerging Trends in Design and Applications of Memory-Based Computing and Content-Addressable Memories</article-title>. <source>Proc. IEEE</source> <volume>103</volume>, <fpage>1311</fpage>&#x2013;<lpage>1330</lpage>. <pub-id pub-id-type="doi">10.1109/jproc.2015.2434888</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kazemi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sahay</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Saxena</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sharifi</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Niemier</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sharon Hu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2021a</year>). <article-title>A Flash-Based Multi-Bit Content-Addressable Memory with Euclidean Squared Distance</article-title>. </citation>
</ref>
<ref id="B25">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kazemi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sharifi</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Laguna</surname>
<given-names>A. F.</given-names>
</name>
<name>
<surname>M&#xfc;ller</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Rajaei</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Olivo</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <source>Memory Nearest Neighbor Search with FeFET Multi-Bit Content-Addressable Memories</source>. </citation>
</ref>
<ref id="B26">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kazemi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sharifi</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Niemier</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X. S.</given-names>
</name>
<name>
<surname>Imani</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2021b</year>). &#x201c;<article-title>MIMHD: Accurate and Efficient Hyperdimensional Inference Using Multi-Bit In-Memory Computing</article-title>,&#x201d; in <source>2021 IEEE/ACM International Symposium on Low Power Electronics and Design (ISLPED)</source>, <fpage>1</fpage>&#x2013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1109/islped52811.2021.9502498</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kitaev</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Kaiser</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Levskaya</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Reformer: The Efficient Transformer</article-title>,&#x201d; in <source>8th International Conference on Learning Representations</source>. </citation>
</ref>
<ref id="B28">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kohonen</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2012</year>). <source>Associative Memory: A System-Theoretical Approach</source>, <volume>17</volume>. <publisher-name>Springer Science &#x26; Business Media</publisher-name>. </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Laguna</surname>
<given-names>A. F.</given-names>
</name>
<name>
<surname>Gamaarachchi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yin</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Niemier</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Parameswaran</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X. S.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Seed-and-Vote Based In-Memory Accelerator for DNA Read Mapping</article-title>,&#x201d; in <source>2020 IEEE/ACM International Conference on Computer Aided Design</source>, <fpage>1</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1145/3400302.3415651</pub-id>
<source>(ICCAD)</source> </citation>
</ref>
<ref id="B30">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Laguna</surname>
<given-names>A. F.</given-names>
</name>
<name>
<surname>Yin</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Reis</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Niemier</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X. S.</given-names>
</name>
</person-group> (<year>2019b</year>). &#x201c;<article-title>Ferroelectric FET Based In-Memory Computing for Few-Shot Learning</article-title>,&#x201d; in <source>Proceedings of the 2019 on Great Lakes Symposium on VLSI</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Homayoun</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Taskin</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Mohsenin</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>W.</given-names>
</name>
</person-group> (<publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>373</fpage>&#x2013;<lpage>378</lpage>. <comment>GLSVLSI &#x2019;19</comment>. <pub-id pub-id-type="doi">10.1145/3299874.3319450</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Laguna</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Niemier</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X. S.</given-names>
</name>
</person-group> (<year>2019a</year>). &#x201c;<article-title>Design of Hardware-Friendly Memory Enhanced Neural Networks</article-title>,&#x201d; in <source>Design, Automation Test in Europe Conference Exhibition (DATE), 2017</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Teich</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Fummi</surname>
<given-names>F.</given-names>
</name>
</person-group>, <fpage>1583</fpage>&#x2013;<lpage>1586</lpage>. <comment>ieeexplore.ieee.org</comment>. <pub-id pub-id-type="doi">10.23919/date.2019.8715198</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lan</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Goodman</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gimpel</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Sharma</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Soricut</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>ALBERT: A Lite BERT for Self-Supervised Learning of Language Representations</article-title>,&#x201d; in <source>ICLR</source>. </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lastras-Monta&#xf1;o</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Del Pozo-Zamudio</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Glebsky</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>K.-T.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Ratio-based Multi-Level Resistive Memory Cells</article-title>. <source>Sci. Rep.</source> <volume>11</volume>, <fpage>1351</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-020-80121-7</pub-id> </citation>
</ref>
<ref id="B34">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lewis</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Goyal</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Ghazvininejad</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mohamed</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Levy</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <source>BART: Denoising Sequence-To-Sequence Pre-training for Natural Language Generation, Translation, and Comprehension</source>. </citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Graves</surname>
<given-names>C. E.</given-names>
</name>
<name>
<surname>Sheng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Miller</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Foltin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pedretti</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2020a</year>). <article-title>Analog Content-Addressable Memories with Memristors</article-title>. <source>Nat. Commun.</source> <volume>11</volume>, <fpage>1638</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-020-15254-4</pub-id> </citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>W.-C.</given-names>
</name>
<name>
<surname>Levy</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.-H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>P.-H.</given-names>
</name>
<etal/>
</person-group> (<year>2021a</year>). <article-title>SAPIENS: A 64-kb RRAM-Based Non-volatile Associative Memory for One-Shot Learning and Inference at the Edge</article-title>. <source>IEEE Trans. Electron. Devices</source> <volume>68</volume>, <fpage>6637</fpage>&#x2013;<lpage>6643</lpage>. <pub-id pub-id-type="doi">10.1109/ted.2021.3110464</pub-id> </citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2021b</year>). <article-title>Across-Array Coding for Resistive Memories with Sneak-Path Interference and Lognormal Distributed Resistance Variations</article-title>. <source>IEEE Commun. Lett.</source> <volume>25</volume>, <fpage>3458</fpage>&#x2013;<lpage>3462</lpage>. <pub-id pub-id-type="doi">10.1109/lcomm.2021.3111218</pub-id> </citation>
</ref>
<ref id="B38">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Niu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020b</year>). <source>Bridging Text and Video: A Universal Multimodal Transformer for Video-Audio Scene-Aware Dialog</source>. </citation>
</ref>
<ref id="B39">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wallace</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Keutzer</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Klein</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2020c</year>). &#x201c;<article-title>Train Big, Then Compress: Rethinking Model Size for Efficient Training and Inference of Transformers</article-title>,&#x201d; In <source>Proceedings of the 37th International Conference on Machine Learning</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Iii</surname>
<given-names>H. D.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>A.</given-names>
</name>
</person-group>, <fpage>5958</fpage>&#x2013;<lpage>5968</lpage>. <comment>(PMLR), vol. 119 of Proceedings of Machine Learning Research</comment>. </citation>
</ref>
<ref id="B40">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <source>Swin Transformer: Hierarchical Vision Transformer Using Shifted Windows</source>. </citation>
</ref>
<ref id="B41">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Marcus</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Santorini</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Marcinkiewicz</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>1993</year>). <source>Building a Large Annotated Corpus of English: The Penn Treebank</source>. </citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Merity</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bradbury</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Socher</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Pointer sentinel Mixture Models</article-title> </citation>
</ref>
<ref id="B43">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Mutlu</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Ghose</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>G&#xf3;mez-Luna</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ausavarungnirun</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2020</year>). <source>A Modern Primer on Processing in Memory</source>. </citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ni</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Yin</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Laguna</surname>
<given-names>A. F.</given-names>
</name>
<name>
<surname>Joshi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>D&#xfc;nkel</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Trentzsch</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Ferroelectric Ternary Content-Addressable Memory for One-Shot Learning</article-title>. <source>Nat. Electron.</source> <volume>2</volume>, <fpage>521</fpage>&#x2013;<lpage>529</lpage>. <pub-id pub-id-type="doi">10.1038/s41928-019-0321-3</pub-id> </citation>
</ref>
<ref id="B45">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Prato</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Charlaix</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Rezagholizadeh</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2019</year>). <source>Fully Quantized Transformer for Machine Translation</source>. </citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Radford</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Child</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Luan</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Amodei</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Sutskever</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Language Models Are Unsupervised Multitask Learners</article-title>. <source>OpenAI blog</source> <volume>1</volume>, <fpage>9</fpage>. </citation>
</ref>
<ref id="B47">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Rae</surname>
<given-names>J. W.</given-names>
</name>
<name>
<surname>Potapenko</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Jayakumar</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Lillicrap</surname>
<given-names>T. P.</given-names>
</name>
</person-group> (<year>2019</year>). <source>Compressive Transformers for Long-Range Sequence Modelling</source>. </citation>
</ref>
<ref id="B48">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Raffel</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Shazeer</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Roberts</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Narang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Matena</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <source>Exploring the Limits of Transfer Learning with a Unified Text-To-Text Transformer</source>, <fpage>10683</fpage>. <comment>arXiv preprint arXiv:1910</comment>. </citation>
</ref>
<ref id="B49">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ranjan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Jain</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Stevens</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Das</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Kaul</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Raghunathan</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>X-MANN: A Crossbar Based Architecture for Memory Augmented Neural Networks</article-title>,&#x201d; in <source>Proceedings of the 56th Annual Design Automation Conference 2019</source> (<publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>6</lpage>. <comment>Article 130 in DAC &#x2019;19</comment>. </citation>
</ref>
<ref id="B50">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Reis</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Laguna</surname>
<given-names>A. F.</given-names>
</name>
<name>
<surname>Niemier</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X. S.</given-names>
</name>
</person-group> (<year>2020a</year>). &#x201c;<article-title>A Fast and Energy Efficient Computing-In-Memory Architecture for Few-Shot Learning Applications</article-title>,&#x201d; in <source>2020 Design, Automation Test in Europe Conference Exhibition (DATE)</source>, <fpage>127</fpage>&#x2013;<lpage>132</lpage>. <comment>ieeexplore.ieee.org</comment>. <pub-id pub-id-type="doi">10.23919/date48585.2020.9116292</pub-id> </citation>
</ref>
<ref id="B51">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Reis</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Laguna</surname>
<given-names>A. F.</given-names>
</name>
<name>
<surname>Niemier</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X. S.</given-names>
</name>
</person-group> (<year>2021</year>).<article-title>Attention-in-Memory for Few-Shot Learning with Configurable Ferroelectric FET Arrays</article-title>. In <conf-name>Proceedings of the 26th Asia and South Pacific Design Automation Conference</conf-name>. <publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>, <fpage>49</fpage>&#x2013;<lpage>54</lpage>. <comment>ASPDAC &#x2019;21</comment>. <pub-id pub-id-type="doi">10.1145/3394885.3431526</pub-id> </citation>
</ref>
<ref id="B52">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Reis</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Niemier</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X. S.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Computing in Memory with FeFETs</article-title>,&#x201d; in <source>Proceedings of the International Symposium on Low Power Electronics and Design</source> (<publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>ACM</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>6</lpage>. <comment>Article 24 in ISLPED &#x2019;18</comment>. <pub-id pub-id-type="doi">10.1145/3218603.3218640</pub-id> </citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reis</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Takeshita</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jung</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Niemier</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X. S.</given-names>
</name>
</person-group> (<year>2020b</year>). <article-title>Computing-in-Memory for Performance and Energy-Efficient Homomorphic Encryption</article-title>. <source>IEEE Trans. VLSI Syst.</source> <volume>28</volume>, <fpage>2300</fpage>&#x2013;<lpage>2313</lpage>. <pub-id pub-id-type="doi">10.1109/tvlsi.2020.3017595</pub-id> </citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Roy</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Saffar</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Vaswani</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Grangier</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Efficient Content-Based Sparse Attention with Routing Transformers</article-title>. <source>Trans. Assoc. Comput. Linguistics</source> <volume>9</volume>, <fpage>53</fpage>&#x2013;<lpage>68</lpage>. <pub-id pub-id-type="doi">10.1162/tacl_a_00353</pub-id> </citation>
</ref>
<ref id="B55">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Roy</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Chakraborty</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Ali</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ankit</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Agrawal</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>In-Memory Computing in Emerging Memory Technologies for Machine Learning: An Overview</article-title>,&#x201d; in <source>2020 57th ACM/IEEE Design Automation Conference (DAC)</source>, <fpage>1</fpage>&#x2013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1109/dac18072.2020.9218505</pub-id> </citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sebastian</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Le Gallo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Khaddam-Aljameh</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Eleftheriou</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Memory Devices and Applications for In-Memory Computing</article-title>. <source>Nat. Nanotechnol.</source> <volume>15</volume>, <fpage>529</fpage>&#x2013;<lpage>544</lpage>. <pub-id pub-id-type="doi">10.1038/s41565-020-0655-z</pub-id> </citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shafiee</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Nag</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Muralimanohar</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Balasubramonian</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Strachan</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>).<article-title>Isaac</article-title>. <source>SIGARCH Comput. Archit. News</source> <volume>44</volume>, <fpage>14</fpage>&#x2013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1145/3007787.3001139</pub-id> </citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sharifi</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Pentecost</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Rajaei</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Kazemi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lou</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>G.-Y.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Application-driven Design Exploration for Dense Ferroelectric Embedded Non-volatile Memories</article-title>. </citation>
</ref>
<ref id="B59">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sharir</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Peleg</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Shoham</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <source>The Cost of Training NLP Models: A Concise Overview</source>. </citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shoeybi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Patwary</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Puri</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>LeGresley</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Casper</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Catanzaro</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism</article-title> </citation>
</ref>
<ref id="B61">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tay</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bahri</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Metzler</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Juan</surname>
<given-names>D.-C.</given-names>
</name>
</person-group> (<year>2020a</year>). &#x201c;<article-title>Sparse Sinkhorn Attention</article-title>,&#x201d; in <source>International Conference on Machine Learning</source>, <fpage>9438</fpage>&#x2013;<lpage>9447</lpage>. </citation>
</ref>
<ref id="B62">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tay</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Dehghani</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Abnar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bahri</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Pham</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2020b</year>). <source>Long Range arena: A Benchmark for Efficient Transformers</source>. </citation>
</ref>
<ref id="B63">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Vaswani</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shazeer</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Parmar</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Uszkoreit</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gomez</surname>
<given-names>A. N.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). &#x201c;<article-title>Attention Is All You Need</article-title>,&#x201d;. In <source>Advances in Neural Information Processing Systems</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Guyon</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Luxburg</surname>
<given-names>U. V.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wallach</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Fergus</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Vishwanathan</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<publisher-loc>Long Beach, California</publisher-loc>: <publisher-name>Curran Associates, Inc.</publisher-name>), <volume>30</volume>, <fpage>5998</fpage>&#x2013;<lpage>6008</lpage>. </citation>
</ref>
<ref id="B64">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Michael</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hill</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Levy</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Bowman</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>GLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding</article-title>,&#x201d; in <source>Proceedings of the 2018 EMNLP Workshop BlackboxNLP: Analyzing and Interpreting Neural Networks for NLP</source> (<publisher-loc>Brussels, Belgium</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>353</fpage>&#x2013;<lpage>355</lpage>. <pub-id pub-id-type="doi">10.18653/v1/w18-5446</pub-id> </citation>
</ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hou</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Attentive Fusion Enhanced Audio-Visual Encoding for Transformer Based Robust Speech Recognition</article-title>. </citation>
</ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>S.-W.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>A. T.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>H.-Y.</given-names>
</name>
</person-group> (<year>2020a</year>). <article-title>Understanding Self-Attention of Self-Supervised Audio Transformers</article-title>. </citation>
</ref>
<ref id="B67">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2020b</year>). &#x201c;<article-title>ReTransformer: ReRAM-Based Processing-In-Memory Architecture for Transformer Acceleration</article-title>,&#x201d; in <conf-name>Proceedings of the 39th International Conference on Computer-Aided Design</conf-name> (<publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>9</lpage>. <comment>Article 92 in ICCAD &#x2019;20</comment>. </citation>
</ref>
<ref id="B68">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Carbonell</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Salakhutdinov</surname>
<given-names>R. R.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>Q. V.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>XLNet: Generalized Autoregressive Pretraining for Language Understanding</article-title>
<source>Adv. Neural Inf. Process. Syst.</source> (<publisher-loc>Vancouver, Canada</publisher-loc>: <publisher-name>Curran Associates, Inc.</publisher-name>) <volume>32</volume>, <fpage>5753</fpage>&#x2013;<lpage>5763</lpage>. </citation>
</ref>
<ref id="B69">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yen-Jen Chang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>A High-Performance and Energy-Efficient TCAM Design for IP-Address Lookup</article-title>. <source>IEEE Trans. Circuits Syst.</source> <volume>56</volume>, <fpage>479</fpage>&#x2013;<lpage>483</lpage>. <pub-id pub-id-type="doi">10.1109/tcsii.2009.2020935</pub-id> </citation>
</ref>
<ref id="B70">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yin</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Niemier</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X. S.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>FeCAM: A Universal Compact Digital and Analog Content Addressable Memory Using Ferroelectric</article-title>. <source>IEEE Trans. Electron. Devices</source> <volume>67</volume>, <fpage>2785</fpage>&#x2013;<lpage>2792</lpage>. <pub-id pub-id-type="doi">10.1109/ted.2020.2994896</pub-id> </citation>
</ref>
<ref id="B71">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yin</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ni</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Reis</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Datta</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Niemier</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X. S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>An Ultra-dense 2FeFET TCAM Design Based on a Multi-Domain FeFET Model</article-title>. <source>IEEE Trans. Circuits Syst.</source> <volume>66</volume>, <fpage>1577</fpage>&#x2013;<lpage>1581</lpage>. <pub-id pub-id-type="doi">10.1109/tcsii.2018.2889225</pub-id> </citation>
</ref>
<ref id="B72">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Yin</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Niemier</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X. S.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Design and Benchmarking of Ferroelectric FET Based TCAM</article-title>,&#x201d; in <source>Design, Automation Test in Europe Conference Exhibition (DATE)</source> (<publisher-loc>Lausanne, Switzerland</publisher-loc>), <fpage>1444</fpage>&#x2013;<lpage>1449</lpage>. <pub-id pub-id-type="doi">10.23919/date.2017.7927219</pub-id> </citation>
</ref>
<ref id="B73">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>P.-Y.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Emerging Memory Technologies: Recent Trends and Prospects</article-title>. <source>IEEE Solid-state Circuits Mag.</source> <volume>8</volume>, <fpage>43</fpage>&#x2013;<lpage>56</lpage>. <pub-id pub-id-type="doi">10.1109/mssc.2016.2546199</pub-id> </citation>
</ref>
<ref id="B74">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zafrir</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Boudoukh</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Izsak</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Wasserblat</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2019</year>). <source>Q8BERT: Quantized 8bit BERT</source>. </citation>
</ref>
<ref id="B75">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zaheer</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Guruganesh</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Dubey</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Ainslie</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Alberti</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ontanon</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). &#x201c;<article-title>Big Bird: Transformers for Longer Sequences</article-title>,&#x201d; in <source>NeurIPS</source>. </citation>
</ref>
<ref id="B76">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Verma</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>In-Memory Computation of a Machine-Learning Classifier in a Standard 6T SRAM Array</article-title>. <source>IEEE J. Solid-state Circuits</source> <volume>52</volume>, <fpage>915</fpage>&#x2013;<lpage>924</lpage>. <pub-id pub-id-type="doi">10.1109/jssc.2016.2642198</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>