<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2024.1268317</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>LRMP: Layer Replication with Mixed Precision for spatial in-memory DNN accelerators</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Nallathambi</surname> <given-names>Abinand</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1298647/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Bose</surname> <given-names>Christin David</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2410036/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Haensch</surname> <given-names>Wilfried</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2619381/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Raghunathan</surname> <given-names>Anand</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Elmore Family School of Electrical and Computer Engineering, Purdue University</institution>, <addr-line>West Lafayette, IN</addr-line>, <country>United States</country></aff>
<aff id="aff2"><sup>2</sup><institution>Argonne National Laboratory, Materials Science Division</institution>, <addr-line>IL</addr-line>, <country>United States</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Rashid Mehmood, King Abdulaziz University, Saudi Arabia</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Jaros&#x00142;aw Pawe&#x00142; Drapa&#x00142;a, Wroc&#x00142;aw University of Technology, Poland</p>
<p>Jorge Daniel Aguirre Morales, UMR7334 Institut des Mat&#x000E9;riaux, de Micro&#x000E9;lectronique et des Nanosciences de Provence (IM2NP), France</p>
<p>Sapan Agarwal, Sandia National Laboratories, United States</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Abinand Nallathambi <email>anallath&#x00040;purdue.edu</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>04</day>
<month>10</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>7</volume>
<elocation-id>1268317</elocation-id>
<history>
<date date-type="received">
<day>27</day>
<month>07</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>17</day>
<month>06</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2024 Nallathambi, Bose, Haensch and Raghunathan.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Nallathambi, Bose, Haensch and Raghunathan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>In-memory computing (IMC) with non-volatile memories (NVMs) has emerged as a promising approach to address the rapidly growing computational demands of Deep Neural Networks (DNNs). Mapping DNN layers spatially onto NVM-based IMC accelerators achieves high degrees of parallelism. However, two challenges that arise in this approach are the highly non-uniform distribution of layer processing times and high area requirements. We propose LRMP, a method to jointly apply layer replication and mixed precision quantization to improve the performance of DNNs when mapped to area-constrained IMC accelerators. LRMP uses a combination of reinforcement learning and mixed integer linear programming to search the replication-quantization design space using a model that is closely informed by the target hardware architecture. Across five DNN benchmarks, LRMP achieves 2.6&#x02013;9.3&#x000D7; latency and 8&#x02013;18&#x000D7; throughput improvement at minimal (&#x0003C;1%) degradation in accuracy.</p></abstract>
<kwd-group>
<kwd>in-memory computing</kwd>
<kwd>analog accelerator</kwd>
<kwd>quantization</kwd>
<kwd>reinforcement learning</kwd>
<kwd>mixed integer linear programming</kwd>
</kwd-group>
<counts>
<fig-count count="9"/>
<table-count count="2"/>
<equation-count count="9"/>
<ref-count count="44"/>
<page-count count="13"/>
<word-count count="8267"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Machine Learning and Artificial Intelligence</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Deep Neural Networks (DNNs) have come to dominate the field of machine learning, and achieve state-of-the-art performance on a variety of complex tasks. However, the advancements in their capabilities have come at the cost of a steep growth in model size and computational complexity. Researchers have developed specialized digital accelerator architectures (Chen et al., <xref ref-type="bibr" rid="B6">2016</xref>; Jouppi et al., <xref ref-type="bibr" rid="B18">2018</xref>; Lie, <xref ref-type="bibr" rid="B26">2022</xref>) and methodologies like quantization and pruning (Liang et al., <xref ref-type="bibr" rid="B25">2021</xref>) that attempt to strike better tradeoffs between cost and functional performance. These implementations are, however, fundamentally limited by the memory bottleneck, as memory accesses are significantly more expensive than arithmetic operations.</p>
<p>In-Memory Computing (IMC) is a computing paradigm where the elementary operations of input vector-weight matrix multiplications in DNNs are performed within memory arrays, potentially alleviating the memory bottleneck. IMC systems have been designed and prototyped with various memory technologies, including SRAM (Zhang et al., <xref ref-type="bibr" rid="B43">2017</xref>; Kang et al., <xref ref-type="bibr" rid="B20">2020</xref>; Yin et al., <xref ref-type="bibr" rid="B42">2020</xref>), DRAM (Gao et al., <xref ref-type="bibr" rid="B9">2019</xref>), and emerging non-volatile memories (NVM) such as RRAM (Chi et al., <xref ref-type="bibr" rid="B7">2016</xref>; Shafiee et al., <xref ref-type="bibr" rid="B37">2016</xref>; Song et al., <xref ref-type="bibr" rid="B39">2017</xref>), PCM (Burr et al., <xref ref-type="bibr" rid="B3">2015</xref>; Khaddam-Aljameh et al., <xref ref-type="bibr" rid="B21">2021</xref>; Narayanan et al., <xref ref-type="bibr" rid="B30">2021</xref>) and STT-MRAM (Jain et al., <xref ref-type="bibr" rid="B14">2018</xref>; Yan et al., <xref ref-type="bibr" rid="B41">2018</xref>). In this work, we focus on IMC with emerging NVMs, where the weights are programmed into the NVM arrays as the conductance, of the memory device. To perform a vector-matrix multiplication (VMM) using such a memory array, the input vector can be presented simultaneously along multiple wordlines using digital-to-analog converters (DACs). The current that flows through each memory cell is the product of the conductance of the memory element (weight) and the wordline voltage (input). These currents naturally sum up at each bit line. These behaviors are dictated by Ohm&#x00027;s and Kirchoff&#x00027;s laws. These bit line currents can then be digitized using analog to digital converters (ADCs) to produce the input vector-weight matrix dot product.</p>
<p>Emerging non-volatile memories have a lot of desirable qualities that make them a good fit for in-memory computing. Their high density means that larger models can be stored, and their non-volatility eliminates the need for continuous power or refresh. However, their high programming costs coupled with their limited endurance make frequent weight re-programming undesirable. These factors make NVMs suitable for weight-stationary inference architectures where all the weights are programmed spatially across the chip and activations flow through and get processed by the appropriate arrays. A consequence of this approach is that the required area scales with the size of the network. While NVM arrays are compact, the peripherals required for IMC (ADCs and DACs) can be quite large, lowering the effective density (storage capacity per unit area) and thus, resulting in large area requirements. To mitigate the large area requirements, researchers have proposed bit-decomposed architectures (Shafiee et al., <xref ref-type="bibr" rid="B37">2016</xref>; Ankit et al., <xref ref-type="bibr" rid="B1">2019</xref>), in which weight bits are stored in spatially distinct arrays and input bits are processed serially, reducing the precision requirements of ADCs and DACs. As the area of the peripherals scale with their precision, the reduced precisions of ADCs and DACs in bit-decomposed architectures result in better effective density, thereby lowering area requirements of IMC accelerators.</p>
<p>While mitigating the area requirements is important, the system performance is also an important consideration. Despite the impressive peak performance offered by spatial IMC accelerators, the actual performance achieved can be significantly lower due to poor utilization caused by the non-uniformity in processing times across the layers of a DNN. Effective mapping techniques can help close the gap between peak and actual performance (Jain et al., <xref ref-type="bibr" rid="B16">2023</xref>). <italic>Layer replication</italic>, which replicates bottleneck layers to facilitate tensor and data parallelism, can balance layer processing times and thus, improve performance. However, layer replication is not a trivial optimization. Finding the resources to replicate the layers, choosing the right layers to replicate, and the number of times to replicate the chosen layers, are all decisions that involve complex tradeoffs between area requirements and performance. Also, with growing model sizes, improving the performance of spatial architectures with layer replication becomes challenging since it exacerbates the area requirements.</p>
<p>Researchers have proposed various techniques to address the challenges of IMC accelerator design. For example, various mixed precision quantization techniques that assign specialized bitwidths to the weights and activations across the layers of a DNN using different optimization strategies (Huang et al., <xref ref-type="bibr" rid="B12">2021</xref>; Kang et al., <xref ref-type="bibr" rid="B19">2021</xref>; Meng et al., <xref ref-type="bibr" rid="B29">2021b</xref>; Peng et al., <xref ref-type="bibr" rid="B31">2022</xref>) have been shown to achieve significant compression of weight and activation footprints, resulting in area, speed and energy improvements. However, these techniques do not address the severe under-utlization of NVM tiles caused by the imbalance in processing times across the layers of DNNs. Others have proposed layer replication techniques to improve utilization that either do not address the question of where to find the resources to replicate layers (Rasch et al., <xref ref-type="bibr" rid="B34">2019</xref>; Li et al., <xref ref-type="bibr" rid="B22">2020</xref>; Li W. et al., <xref ref-type="bibr" rid="B23">2023</xref>) or rely on design-time tradeoffs to accommodate the replicated layers (He et al., <xref ref-type="bibr" rid="B11">2022</xref>). In contrast, we identify a novel synergy between <bold>L</bold>ayer <bold>R</bold>eplication and <bold>M</bold>ixed <bold>P</bold>recision quantization that can be exploited at compilation-time to improve the performance of DNNs on spatial IMC accelerators. We propose LRMP, an automated mapping framework that explores the quantization-replication design space to quantize layers selectively to free up resources and then replicate the right layers using the freed-up resources to improve performance. In summary, our contributions are as follows:</p>
<list list-type="bullet">
<list-item><p>We present LRMP, a novel framework that jointly performs mixed precision quantization and layer replication during mapping of DNNs to IMC accelerators.</p></list-item>
<list-item><p>We propose a joint-optimization approach with (i) deep reinforcement learning based selection of precision for each layer in the network to maintain accuracy while conserving hardware resources, and (ii) linear programming based selective layer replication to redeploy the conserved resources in the IMC hardware to improve performance.</p></list-item>
<list-item><p>We evaluate the LRMP framework on a benchmark suite of convolutional and fully-connected neural networks and achieve 2.6&#x02013;9.3&#x000D7; latency improvement and 8&#x02013;18&#x000D7; throughput improvement, at iso-area and near iso-accuracy.</p></list-item>
</list>
<p>The rest of the paper is organized as follows. We describe the process of mapping a neural network layer in an IMC system and discuss the implications of precision on resource requirements and latency in Section 2. We motivate the synergy between mixed precision and layer replication using an illustrated example in Section 3. We present the details of the LRMP framework in Section 4. We describe our experimental setup in Section 5 and present our results in Section 6. We discuss the contributions of our work in the context of existing related works in Section 7. Finally, we conclude the paper in Section 8.</p>
</sec>
<sec id="s2">
<title>2 Preliminaries</title>
<p>Vector-matrix multiplication (VMM) is an elementary operation in the evaluation of neural networks. In this section, we describe how a weight matrix can be mapped to multiple crossbar tiles and how these crossbar tiles can collectively perform a multiplication operation between an input vector and a weight matrix to produce an output vector. We also describe the latency and the number of crossbar tiles required for such an implementation of VMM.</p>
<p>Convolutional layers represent a common layer configuration used in DNNs. They are composed of a three-dimensional weight tensor array sliding across a three-dimensional input tensor, producing an output value for each patch of overlap. Convolutional layers are realized on IMC substrates by converting the weight tensor into a two-dimensional matrix and performing image-to-column lowering of the input tensor into a sequence of vectors. Then, the convolution output values can be produced by performing a sequence of VMMs.</p>
<p>Consider a convolution operation with <italic>C</italic> input features, <italic>N</italic> output features and a kernel size of <italic>K</italic>, producing output features of dimension <italic>W</italic> &#x000D7; <italic>W</italic>. The size of its lowered weight matrix is <italic>K</italic><sup>2</sup><italic>C</italic> &#x000D7; <italic>N</italic>. The input tensor is transformed into <italic>W</italic><sup>2</sup> vectors of length <italic>K</italic><sup>2</sup><italic>C</italic>. With a sequence of VMMs, <italic>W</italic><sup>2</sup> output vectors of length <italic>N</italic> are produced. It must be noted that the number of vectors can be quite high and it depends on the dimensions of the input, the filter kernel size <italic>K</italic>, the padding and the stride. For, example, in the first convolutional layer of the ResNet18 DNN, the input matrix has over 12,000 column vectors.</p>
<p>To map a convolutional layer to a crossbar, the weight tensor is first lowered to a two-dimensional matrix, as shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. The weight matrix is then segmented into multiple sub-matrices of size <italic>X</italic> &#x000D7; <italic>X</italic>, which denotes the size of the crossbar array or tile. To build a spatial architecture, each of these sub-matrices are mapped onto individual crossbar tiles. The number of tiles required to perform this spatial mapping is given by <xref ref-type="disp-formula" rid="E1">Equation 1</xref>.</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mo>&#x00023;</mml:mo><mml:mtext>tiles</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:mi>K</mml:mi><mml:mo>,</mml:mo><mml:mi>C</mml:mi><mml:mo>,</mml:mo><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mi>X</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mo>&#x02308;</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msup><mml:mi>K</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mi>C</mml:mi></mml:mrow><mml:mi>X</mml:mi></mml:mfrac></mml:mrow><mml:mo>&#x02309;</mml:mo></mml:mrow><mml:mo>&#x000D7;</mml:mo><mml:mrow><mml:mo>&#x02308;</mml:mo><mml:mrow><mml:mfrac><mml:mi>N</mml:mi><mml:mi>X</mml:mi></mml:mfrac></mml:mrow><mml:mo>&#x02309;</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Realizing a VMM operation using crossbar arrays.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1268317-g0001.tif"/>
</fig>
<p>As discussed in Section 1, crossbar arrays can be built with a variety of memory technologies. Also, these memory devices are designed to store a specific number of bits. The precision of the memory element is a matter of concern, as high precision elements have been shown to be more sensitive to process variations and conductance drift (Shim et al., <xref ref-type="bibr" rid="B38">2021</xref>) and incur higher programming costs (Perez et al., <xref ref-type="bibr" rid="B33">2021</xref>). Thus, low precision devices are desirable with regards to both accuracy and performance.</p>
<p>The achievable precision of the memory device needs to be reconciled with the required logical precision of the weights. If the device precision (<italic>s</italic><sub><italic>b</italic></sub>) is less than the required weight precision (<italic>w</italic><sub><italic>b</italic></sub>), the weight sub-matrices can be sliced into groups of <italic>s</italic><sub><italic>b</italic></sub> bits and then each slice can be mapped to a separate crossbar tile, as shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. It must be noted that the digital outputs corresponding to the bit-slices of the weight matrix need to be appropriately shifted and added to produce the final output. The number of tiles required considering a bit-sliced mapping is given by <xref ref-type="disp-formula" rid="E2">Equation 2</xref>.</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mo>&#x00023;</mml:mo><mml:mtext>tiles</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:mi>K</mml:mi><mml:mo>,</mml:mo><mml:mi>C</mml:mi><mml:mo>,</mml:mo><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mi>X</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mi>b</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>s</mml:mi><mml:mi>b</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mo>&#x02308;</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msup><mml:mi>K</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mi>C</mml:mi></mml:mrow><mml:mi>X</mml:mi></mml:mfrac></mml:mrow><mml:mo>&#x02309;</mml:mo></mml:mrow><mml:mo>&#x000D7;</mml:mo><mml:mrow><mml:mo>&#x02308;</mml:mo><mml:mrow><mml:mfrac><mml:mi>N</mml:mi><mml:mi>X</mml:mi></mml:mfrac></mml:mrow><mml:mo>&#x02309;</mml:mo></mml:mrow><mml:mo>&#x000D7;</mml:mo><mml:mrow><mml:mo>&#x02308;</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mi>b</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mi>b</mml:mi></mml:msub></mml:mrow></mml:mfrac></mml:mrow><mml:mo>&#x02309;</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>As discussed in Section 1, to perform a vector-matrix multiplication using crossbars, we must convert the input vector values to analog voltages using a digital-to-analog converter (DAC). The output currents of the crossbar array are then converted to digital using analog-to-digital converters (ADC). These peripheral circuits, especially ADCs, occupy a significant proportion of latency and, area and power budgets. The design choices of the number of ADCs per crossbar array, and ADC and DAC precisions are important considerations in the design of crossbar-based architectures. The number of ADCs can be chosen to be lesser than the number of columns and the ADCs can be time-multiplexed between multiple columns. In order to reduce the precision requirements of ADCs and DACs, we can stream the input vectors bit-by-bit and reduce the corresponding outputs with shift-add operations. The latency of performing VMMs required by a convolution layer in such a bit-streamed manner is given by <xref ref-type="disp-formula" rid="E3">Equation 3</xref>.</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext>lat</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:mi>W</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mi>b</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mi>X</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mi>D</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:msup><mml:mi>W</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:mrow><mml:mo>&#x02308;</mml:mo><mml:mrow><mml:mfrac><mml:mi>X</mml:mi><mml:mrow><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mi>D</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mrow><mml:mo>&#x02309;</mml:mo></mml:mrow><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mi>b</mml:mi></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>n</italic><sub><italic>ADC</italic></sub> is the number of ADCs per crossbar array, <italic>t</italic><sub><italic>tile</italic></sub> is the time elapsed between presenting an input to the tile and the ADCs producing their output, <italic>a</italic><sub><italic>b</italic></sub> is the number of bits required to represent the input vector values and <italic>W</italic><sup>2</sup> is the number of vectors.</p>
<p>Thus, both the hardware requirements and latency of a crossbar-based architecture depend on the precision of the weights and activations of the neural networks mapped onto them.</p>
</sec>
<sec id="s3">
<title>3 Motivation</title>
<p>In this section, we illustrate how mixed precision and layer replication greatly impact latency and throughput of DNN evaluation.</p>
<p>Let us consider the baseline implementation of ResNet18 with 8-bit weights and 8-bit activations. As defined by <xref ref-type="disp-formula" rid="E2">Equation 2</xref>, the tile consumption of each layer in a spatial architecture depends on the size of the weight matrix, the logical weight precision (<italic>w</italic><sub><italic>b</italic></sub>) and the physical memory device precision (<italic>s</italic><sub><italic>b</italic></sub>). As shown in <xref ref-type="fig" rid="F2">Figure 2A</xref>, we observe that different layers of the network have different latencies and tile requirements, as defined by <xref ref-type="disp-formula" rid="E3">Equations 3</xref> and <xref ref-type="disp-formula" rid="E2">2</xref>, respectively. In our evaluations, we use a device precision of 1-bit and a crossbar size of 256&#x000D7;256.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>An experimental illustration of heterogeneous quantization and layer replication using ResNet18. <bold>(A)</bold> 8-bit baseline. <bold>(B)</bold> Selective Quantization. <bold>(C)</bold> Layer Replication.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1268317-g0002.tif"/>
</fig>
<p>By selectively reducing the precision of weights in certain layers, we can take advantage of the bit-sliced implementation and reduce the number of tiles required by that layer. These conserved tiles can be used to replicate bottleneck layers in a neural network to process parts of the layer input in parallel, resulting in a latency reduction. Similarly, by selectively reducing the precision of input vectors, we can reduce the number of bits to be streamed to the crossbar arrays, resulting in proportional reduction of latency.</p>
<p>Let us consider reducing the weight precision of a resource-intensive layer and the input precision of the bottleneck layer to 6-bits. As shown in <xref ref-type="fig" rid="F2">Figure 2B</xref>, we observe that 72 tiles are conserved. In addition, as explained by <xref ref-type="disp-formula" rid="E3">Equation 3</xref>, the latency of the bottleneck layer is reduced resulting in an overall 5.7% latency improvement and 1.33&#x000D7; throughput improvement.</p>
<p>If these newly freed-up tiles are used to naively replicate only the bottleneck layer, we can create 9 more copies of that layer. Thus, 10 input vectors of that layer can be processed in parallel, resulting in 25.5% improvement in total latency and 2.34&#x000D7; improvement in throughput, as illustrated in <xref ref-type="fig" rid="F2">Figure 2C</xref>.</p>
<p>The above example illustrates that a trade-off exists between precision and latency in spatial IMC architectures. A few questions that arise related to this trade-off are:</p>
<sec>
<title>3.1 How to choose the precision of each layer?</title>
<p>When choosing the precision of each layer, we need to consider its impact on the tile consumption, overall latency, and accuracy. The weight precision affects the bit-slicing factor, which is only one of the factors that determines the tile consumption of a layer (<xref ref-type="disp-formula" rid="E2">Equation 2</xref>). The other factors depend on the size of the weight matrix. Thus, it is important to choose the weight precision of each layer in a way that the number of tiles conserved is maximized. Similarly, the activation precision only affects the bit-streaming factor of <xref ref-type="disp-formula" rid="E3">Equation 3</xref>. The other factor is the number of input vectors to be processed. Thus, it is important to choose the activation precision of each layer in a way that the latency is minimized. Moreover, reducing the activation/weight precision of any layer in a neural network has implications for the overall accuracy of the network. Thus, it is important to choose the precision of each layer in a way that the overall accuracy is not compromised.</p>
</sec>
<sec>
<title>3.2 Where to repurpose the conserved tiles?</title>
<p>When choosing the replication factor of a layer, we need to consider its impact on the overall latency and tile consumption. The latency of a layer is a function of the number of input vectors and their precision, both of which can vary across the layers of a neural network. At the same time, the number of tiles required to replicate layers also varies based on the size of their weight matrices. Thus, it is important to choose the replication factor of each layer in a way that the utility of the conserved tiles is maximized, and the overall latency is minimized.</p>
</sec>
</sec>
<sec id="s4">
<title>4 LRMP methodology</title>
<p>In this paper, we propose LRMP (Layer Replication through Mixed Precision), a framework that combines reinforcement learning (RL) and mixed integer linear programming (MILP) to jointly optimize latency/throughput and accuracy of DNNs realized on IMC hardware fabrics.</p>
<p>As shown in <xref ref-type="fig" rid="F3">Figure 3</xref> LRMP is an iterative process with each iteration or episode consisting of two-steps: (1) an RL-agent choosing the precision of each layer in the DNN, and (2) an MILP-based optimizer choosing the replication factors of each layer. After each episode, the latency, throughput and accuracy of the network are evaluated and used to guide the RL-agent. In the remainder of this section, we describe the hardware modeling of an RRAM-based IMC accelerator, and then discuss how linear programming and reinforcement learning can be employed in tandem to quantize and replicate layers to jointly optimize accuracy and performance metrics under a chip capacity constraint.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Overview of the proposed LRMP methodology.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1268317-g0003.tif"/>
</fig>
<sec>
<title>4.1 Hardware model</title>
<p>LRMP is a joint optimization process that is designed to improve accuracy and latency/throughput achieved by DNNs on spatial in-memory accelerators. To estimate the performance metrics of evaluating DNNs on these spatial accelerators, we develop a simple and effective cost model that can be used to estimate latency and throughput of a DNN during the optimization process. The cost model is based on the compute-in-memory system developed by Chang et al. (<xref ref-type="bibr" rid="B4">2022</xref>), which consists of a Cortex M3 microprocessor, two vector modules for digital compute, and 288 crossbar tiles of dimension 256&#x000D7;256. Data transport is implemented using 8 lanes of 8-bit wide buses from the vector modules to the crossbar tiles and 8 lanes of 32-bit wide buses from the crossbar tiles to the vector modules. Each vector module has 8 lanes of parallel compute and 128KB of SRAM. The system is equipped with fine-grained power gating and each tile can be individually turned off. On account of the larger computer vision models used to benchmark the proposed approaches in this work, the cost model assumes a scaled-up version of this system with 5688 tiles and 40 vector modules, each with 64 lanes of parallel compute. Further details on the microarchitecture are provided in Section 5.</p>
<p>The latency of evaluating a DNN layer (<italic>T</italic><sub><italic>l</italic></sub>) on this compute-in-memory system, described in <xref ref-type="disp-formula" rid="E4">Equation 4</xref>, comprises of 4 factors: <italic>T</italic><sub><italic>tileIn</italic>.<italic>l</italic></sub>, which refers to the latency of transferring input vectors of layer <italic>l</italic> from vector modules to the respective tiles. This transfer is performed over eight 8-bit lanes, which are shared among 144 tiles. Similarly, <italic>T</italic><sub><italic>tileOut</italic>.<italic>l</italic></sub> represents the latency of transferring output vectors of layer <italic>l</italic> from crossbar tiles to the respective vector modules using eight 32-bit lanes shared by the same 144 tiles. The latency of performing Vector-Matrix-Multiplication (VMM) using crossbar tiles with temporally bit-streamed inputs and spatially bit-sliced weights of layer <italic>l</italic> is represented by <italic>T</italic><sub><italic>tile</italic>.<italic>l</italic></sub>. Lastly, the post-VMM digital compute of layer <italic>l</italic> performed by vector modules has a latency of <italic>T</italic><sub><italic>d</italic>.<italic>l</italic></sub>, which uses 64 lanes processing the output vectors of 144 tiles. It must be noted that each of these components is a function of the number of bits used to represent the input activations and weights of layer <italic>l</italic>.</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>I</mml:mi><mml:mi>n</mml:mi><mml:mo>.</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>O</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi><mml:mo>.</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mo>.</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mo>.</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>With the latency of evaluating a layer defined, the latency of evaluating a DNN is the sum of the latencies of its constituent layers. The latency of evaluating a DNN with <italic>L</italic> layers is given by <xref ref-type="disp-formula" rid="E5">Equation 5</xref>.</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>T</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The system is designed to operate with coarse-grained pipeline parallelism. Thus, the throughput of the system is defined by the maximum latency of any layer. The throughput of the system is given by <xref ref-type="disp-formula" rid="E6">Equation 6</xref>.</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:munder><mml:mrow><mml:mo class="qopname">max</mml:mo></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mtext>&#x000A0;</mml:mtext><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The model described above is used in LRMP to perform analysis and exploration of the quantization and replication design space.</p>
</sec>
<sec>
<title>4.2 Optimizing layer replication using mixed integer linear programming</title>
<p>As discussed in Section 3, tiles can be freed up by selectively quantizing layers based on their tile footprint and selectively replicating layers based on their latencies. When a layer is replicated, the number of tiles and vector modules allocated to that layer is increased. Thus, for the said layer: (1) the total bandwidth available for data transfer is increased; (2) the amount of digital compute allocated is increased; and (3) the number of tiles available for performing the required VMM operations is increased. This results in a linear reduction in the latency of evaluating the layer.</p>
<p>If there are <italic>r</italic><sub><italic>l</italic></sub> instances of layer <italic>l</italic>, then the latency of evaluating the DNN is given by <xref ref-type="disp-formula" rid="E7">Equation 7</xref>.</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>T</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>I</mml:mi><mml:mi>n</mml:mi><mml:mo>.</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>O</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi><mml:mo>.</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mo>.</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mo>.</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Given a mixed precision quantization scheme, the total number of tiles required to have one instance of each layer (<italic>s</italic><sub><italic>l</italic></sub>) is given by <xref ref-type="disp-formula" rid="E2">Equation 2</xref>. The total latency <italic>T</italic> can be optimized by carefully choosing the layer replication factors denoted by the vector <bold>r</bold>. The process of choosing the replication factors is naturally constrained by the total number of tiles available in the system (<italic>N</italic><sub><italic>tiles</italic></sub>). This can be formulated as a constrained optimization problem, as shown in Formulation 8.</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:munder><mml:mrow><mml:mtext>minimize</mml:mtext></mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>r</mml:mi></mml:mstyle></mml:munder><mml:mtext>&#x000A0;&#x000A0;</mml:mtext><mml:mstyle displaystyle='true'><mml:munder><mml:mo>&#x02211;</mml:mo><mml:mi>l</mml:mi></mml:munder><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:msub><mml:mi>r</mml:mi><mml:mi>l</mml:mi></mml:msub></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>I</mml:mi><mml:mi>n</mml:mi><mml:mo>.</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>O</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi><mml:mo>.</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mo>.</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>d</mml:mi><mml:mo>.</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>subject&#x000A0;to</mml:mtext></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mi>r</mml:mi><mml:mi>l</mml:mi></mml:msub><mml:mo>&#x02265;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;</mml:mtext><mml:mstyle displaystyle='true'><mml:munder><mml:mo>&#x02211;</mml:mo><mml:mi>l</mml:mi></mml:munder><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>r</mml:mi><mml:mi>l</mml:mi></mml:msub><mml:mo>*</mml:mo><mml:msub><mml:mi>s</mml:mi><mml:mi>l</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mstyle><mml:mo>&#x02264;</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The constraints ensure that there is at least one instance of each layer and the total number of allocated tiles doesn&#x00027;t exceed the number of tiles available (<italic>N</italic><sub><italic>tiles</italic></sub>). Since <italic>s</italic><sub><italic>l</italic></sub> is constant for a given quantization scheme, the constraints are linear. However, the objective function is non-linear.</p>
<p>As defined by <xref ref-type="disp-formula" rid="E6">Equation 6</xref>, the throughput of the system is the inverse of the maximum latency across all layers. Thus, to maximize throughput, we need to minimize the maximum latency across all layers. Optimizing for throughput is, thus, a min-max problem. The optimization problem can therefore be re-written as shown in Formulation 9.</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M9a"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:munder><mml:mrow><mml:mtext>minimize</mml:mtext></mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>r</mml:mi></mml:mstyle></mml:munder><mml:mtext>&#x000A0;&#x000A0;</mml:mtext><mml:mi>M</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>subject&#x000A0;to</mml:mtext></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:msub><mml:mi>r</mml:mi><mml:mi>l</mml:mi></mml:msub></mml:mrow></mml:mfrac><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>I</mml:mi><mml:mi>n</mml:mi><mml:mo>.</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>O</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi><mml:mo>.</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mo>.</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>d</mml:mi><mml:mo>.</mml:mo><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02264;</mml:mo><mml:mi>M</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>subject&#x000A0;to</mml:mtext></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mi>r</mml:mi><mml:mi>l</mml:mi></mml:msub><mml:mo>&#x02265;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mstyle displaystyle='true'><mml:munder><mml:mo>&#x02211;</mml:mo><mml:mi>l</mml:mi></mml:munder><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>r</mml:mi><mml:mi>l</mml:mi></mml:msub><mml:mo>*</mml:mo><mml:msub><mml:mi>s</mml:mi><mml:mi>l</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mstyle><mml:mo>&#x02264;</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>We introduce a dummy variable <italic>M</italic> and reformulate the optimization problem to minimize <italic>M</italic>, while ensuring that the latency of each layer does not exceed <italic>M</italic>. By constraining the latency of each layer to be no greater than <italic>M</italic> and minimizing <italic>M</italic>, we are effectively minimizing the maximum latency across all layers, which maximizes the throughput of the system.</p>
<p>These optimization problems are not automatically linear. However, we can employ linearization techniques (Asghari et al., <xref ref-type="bibr" rid="B2">2022</xref>) to reformulate the optimization problem with linear constraints and objective function. Then, we solve the reformulated problem using an MILP solver.</p>
</sec>
<sec>
<title>4.3 Constraining the action space with performance budgets</title>
<p>The reinforcement learning framework used in this work is based on the work by Wang et al. (<xref ref-type="bibr" rid="B40">2019</xref>), which imposes a performance cost constraint on the action space of the RL agent. If the quantization policy prescribed by the RL agent does not meet the performance targets, it is modified by decreasing the bitwidths until the performance targets are met. While this approach is effective, it does not provide any insight into the tradeoffs between accuracy and performance. We restructure this approach to explore the tradeoffs between accuracy and performance by exponentially tightening the performance budget. This results in the RL agent exploring the space of quantization policies to not just meet a performance budget but also achieve better performance metrics.</p>
</sec>
<sec>
<title>4.4 Rewarding the RL agent with accuracy and performance metrics</title>
<p>In each episode of exploration, the RL agent is rewarded based on the quality of the quantization policy it prescribes. Wang et al. (<xref ref-type="bibr" rid="B40">2019</xref>) rewarded the RL agent based on the accuracy of the quantized DNN. In this work, we optimize the performance of the quantized DNN by using the layer replication technique. Thus, to achieve joint optimization, the RL agent is rewarded based on the accuracy and performance of the quantized DNN. The reward function is given by <xref ref-type="disp-formula" rid="E10">Equation 10</xref>.</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M9"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mrow><mml:mi mathvariant="script">R</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mi>&#x003BB;</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:msub><mml:mrow><mml:mi>c</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi><mml:mi>u</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:msub><mml:mrow><mml:mi>c</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi><mml:mi>u</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>/</mml:mo><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where, <italic>acc</italic><sub><italic>quant</italic></sub> is the accuracy of the quantized DNN, <italic>acc</italic><sub><italic>original</italic></sub> is the accuracy of the original DNN, <italic>T</italic><sub><italic>quant</italic></sub> is the latency of the quantized DNN and <italic>T</italic><sub><italic>original</italic></sub> is the latency of the original DNN when optimizing for latency. When optimizing for throughput, <italic>T</italic><sub><italic>quant</italic></sub> and <italic>T</italic><sub><italic>original</italic></sub> are latencies of the bottleneck layers of the respective DNNs. The hyperparameters &#x003BB; and &#x003B1; control the relative importance of accuracy and performance in the reward function. The reward function is designed to encourage the RL agent to prescribe quantization policies that result in a quantized DNN that is optimized to balance accuracy and speed.</p>
</sec>
</sec>
<sec id="s5">
<title>5 Experimental methodology</title>
<sec>
<title>5.1 Microarchitectural details</title>
<p>As described in Section 4, this work is based on a scaled-up model of the compute-in-memory system fabricated by Chang et al. (<xref ref-type="bibr" rid="B4">2022</xref>). The microarchitectural parameters are listed in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Microarchitectural parameters.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Parameter</bold></th>
<th valign="top" align="center"><bold>Value</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">eNVM</td>
<td valign="top" align="center">1T-1R RRAM</td>
</tr> <tr>
<td valign="top" align="left">Tile size</td>
<td valign="top" align="center">256&#x000D7; 256</td>
</tr> <tr>
<td valign="top" align="left">No. of tiles</td>
<td valign="top" align="center">5688</td>
</tr> <tr>
<td valign="top" align="left">No. of vector modules</td>
<td valign="top" align="center">40</td>
</tr> <tr>
<td valign="top" align="left">Device precision</td>
<td valign="top" align="center">1 bit</td>
</tr> <tr>
<td valign="top" align="left">Row parallelism</td>
<td valign="top" align="center">9</td>
</tr> <tr>
<td valign="top" align="left">DAC precision</td>
<td valign="top" align="center">1 bit</td>
</tr> <tr>
<td valign="top" align="left">Column parallelism</td>
<td valign="top" align="center">8</td>
</tr> <tr>
<td valign="top" align="left">ADC precision</td>
<td valign="top" align="center">4 bits</td>
</tr> <tr>
<td valign="top" align="left">Avg. power per tile</td>
<td valign="top" align="center">130 &#x003BC;W</td>
</tr>
<tr>
<td valign="top" align="left">Clock frequency</td>
<td valign="top" align="center">192 MHz</td>
</tr></tbody>
</table>
</table-wrap>
<p>The system is built using 1T-1R RRAM eNVM technology, with a tile size of 256x256 and a total of 5688 tiles. The system also includes 40 vector modules, each of which contains 64 lanes of parallel digital compute and 128 KB of SRAM. Each tile is equipped with eight 4-bit Flash ADCs and 256 1-bit DACs. To prevent partial sum quantization and mitigate other non-idealities, only 9 rows are activated at a time. The system is clocked at 192 MHz.</p>
<p>The energy consumption of the system is modeled with three components: power consumed by the RRAM tiles, the energy cost of reading and writing the activations to the on-chip SRAM buffers, and the power leaked by the SRAMs. Each RRAM tile is reported to consume an average power of 130 &#x003BC;W (Chang et al., <xref ref-type="bibr" rid="B4">2022</xref>). The SRAM blocks are modeled using CACTI.</p>
<p>While we evaluate LRMP on a specific architecture, the proposed techniques are not limited to it. The proposed optimizations are applicable to any bit-decomposed IMC architecture (Shafiee et al., <xref ref-type="bibr" rid="B37">2016</xref>; Ankit et al., <xref ref-type="bibr" rid="B1">2019</xref>; Zhu et al., <xref ref-type="bibr" rid="B44">2019</xref>), and are otherwise agnostic to the underlying hardware.</p>
</sec>
<sec>
<title>5.2 Methods</title>
<sec>
<title>5.2.1 Reinforcement learning</title>
<p>As described in Section 4, the reinforcement learning framework used in this work is based on the hardware-aware quantization tool proposed by Wang et al. (<xref ref-type="bibr" rid="B40">2019</xref>). The method consists of two phases: exploration and finetuning. In the exploration phase, the agent explores the action space to find a good policy based on the performance budget and rewards provided, as described in Section 4. The trajectory of the exploration phase is discussed in Section 6.3.</p>
<p>After the exploration phase, the DNN is quantized with the mixed precision scheme found by the agent. In the finetuning phase, the DNN is trained with the quantized weights and activations to recover any accuracy lost to quantization.</p>
</sec>
<sec>
<title>5.2.2 Mixed integer linear programming</title>
<p>Given a quantization policy prescribed by the RL agent, the mixed integer linear programming step is used to find the replication factors that optimize the performance of the system. Optimization objectives of both latency (<italic>latencyOptim</italic>) and throughput (<italic>throughputOptim</italic>) are implemented. The baseline for each network in the benchmark suite is the implementation with 8-bit weights and activations. Thus, the layer replication is performed with a constraint that the total number of tiles used is no more than the baseline. This is a design choice to ensure that performance is optimized without increasing area. An ablation study has been performed and described in Section 6.5 to show the effectiveness of our LRMP method with and without this design constraint.</p>
</sec>
</sec>
<sec>
<title>5.3 Benchmarks</title>
<p>The proposed LRMP approach has been evaluated on a set of DNN benchmarks trained on the ImageNet and MNIST datasets. The baseline of comparison for each benchmark is the implementation with 8-bit weights and activations. The benchmarks are listed in <xref ref-type="table" rid="T2">Table 2</xref>, along with the number of tiles required by the baseline implementation. The multilayer perceptron (MLP) is trained on the MNIST dataset, with 4 hidden layers of 1024, 4096, 4096 and 1024 neurons respectively. The ResNets are finetuned on the ImageNet dataset with pre-trained weights. While we limit our evaluation to DNNs trained for classification tasks, we believe the proposed techniques are broadly applicable to any quantized DNN (Dettmers et al., <xref ref-type="bibr" rid="B8">2022</xref>; Li X. et al., <xref ref-type="bibr" rid="B24">2023</xref>). Also, the proposed techniques do not place any limits on the number of bits used for quantization.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>DNN benchmarks.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Benchmark</bold></th>
<th valign="top" align="center"><bold>Dataset</bold></th>
<th valign="top" align="center"><bold><italic>N</italic><sub><italic>tiles</italic></sub></bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">MLP</td>
<td valign="top" align="center">MNIST</td>
<td valign="top" align="center">3,232</td>
</tr> <tr>
<td valign="top" align="left">ResNet18</td>
<td valign="top" align="center">ImageNet</td>
<td valign="top" align="center">1608</td>
</tr> <tr>
<td valign="top" align="left">ResNet34</td>
<td valign="top" align="center">ImageNet</td>
<td valign="top" align="center">2968</td>
</tr> <tr>
<td valign="top" align="left">ResNet50</td>
<td valign="top" align="center">ImageNet</td>
<td valign="top" align="center">3376</td>
</tr> <tr>
<td valign="top" align="left">ResNet101</td>
<td valign="top" align="center">ImageNet</td>
<td valign="top" align="center">5688</td>
</tr></tbody>
</table>
</table-wrap>
<p>It must be noted that, besides quantization, analog non-idealities such as noise, conductance drift, device-to-device variation etc. have not been modeled in this work. However, modeling these non-idealities (Jain et al., <xref ref-type="bibr" rid="B15">2020</xref>; Lu et al., <xref ref-type="bibr" rid="B27">2021</xref>; Roy et al., <xref ref-type="bibr" rid="B35">2021</xref>) and developing compensation techniques (Charan et al., <xref ref-type="bibr" rid="B5">2020</xref>; Meng et al., <xref ref-type="bibr" rid="B28">2021a</xref>; Jain and Raghunathan, <xref ref-type="bibr" rid="B13">2019</xref>) are areas of active and ongoing research and we believe these effects are not an impediment to the principal contributions of this work.</p>
</sec>
</sec>
<sec sec-type="results" id="s6">
<title>6 Results</title>
<p>In the sub-sections of this section, we first present the latency, throughput and energy improvements achieved by LRMP. We then present results that provide insights into the RL-based exploration process. We also show a layer-wise breakdown of how latencies are optimized and an ablation study that analyses the sensitivity of the layer replication methodology to area constraints.</p>
<sec>
<title>6.1 Latency and throughput improvements</title>
<p><xref ref-type="fig" rid="F4">Figure 4</xref> reports the latency and throughput improvements achieved by the LRMP framework. As explained in Section 5, the improvements are reported with respect to fixed-precision baseline networks with 8-bit weights and activations.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Latency and throughput improvements achieved by LRMP.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1268317-g0004.tif"/>
</fig>
<p>We observe 2.6&#x02013;9.3&#x000D7; reduction in latency and 6.6&#x02013;15&#x000D7; improvement in throughput while optimizing for latency (denoted as <italic>latencyOptim</italic>) across the suite of benchmark DNNs. Similarly, we observe 8&#x02013;18&#x000D7; improvement in throughput and 2.5&#x02013;7.8&#x000D7; reduction in latency while optimizing for throughput (denoted as <italic>throughputOptim</italic>). These improvements are obtained with accuracy loss of less than 1% after finetuning with the quantization policies determined by LRMP.</p>
</sec>
<sec>
<title>6.2 Energy improvements</title>
<p>Although LRMP explicitly optimizes for throughput or latency, it achieves energy improvements as a result of more efficient DNN execution on the IMC substrate. <xref ref-type="fig" rid="F5">Figure 5</xref> shows the energy improvements achieved by LRMP. We observe 4.75&#x02013;8.9&#x000D7; improvement in energy consumption while optimizing for throughput and 4.7&#x02013;8&#x000D7; energy improvement while optimizing for latency.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Energy improvements achieved by LRMP.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1268317-g0005.tif"/>
</fig>
</sec>
<sec>
<title>6.3 Studying joint optimization of accuracy and performance</title>
<p>As discussed in Section 4, the proposed approach jointly optimizes for accuracy and performance by rewarding the RL agent with an affine combination of accuracy and performance metrics, and by continuously tightening the constraints placed on the action space. <xref ref-type="fig" rid="F6">Figure 6</xref> shows the trajectory of the RL agent performing latency optimization for ResNet18. The exploration is started with a lenient performance budget of 0.35&#x000D7; baseline latency and exponentially tightened to 0.2&#x000D7; baseline latency. Over the course of the exploration, the agent finds quantization policies that achieve upto 5&#x000D7; improvement in latency with layer replication while also improving the accuracy.</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Trajectory of RL agent jointly optimizing ResNet18 for accuracy and latency.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1268317-g0006.tif"/>
</fig>
</sec>
<sec>
<title>6.4 Layer-wise breakdown</title>
<p>As discussed in Section 4, the layer replication can be performed by optimizing for either latency or throughput. The two objectives have different implications on the latencies and tile consumptions of each layer and thus, latency and throughput outcomes for the overall network. <xref ref-type="fig" rid="F7">Figure 7</xref> shows the layer-wise breakdown of latencies and tiles for ResNet18 for the baseline implementation as well as the LRMP implementation while optimizing for latency and throughput. <xref ref-type="fig" rid="F8">Figure 8</xref> shows the quantization policies found by LRMP for ResNet18 while optimizing for latency and throughput.</p>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>Layer-wise breakdown of latencies and tiles for ResNet18 for the baseline and while optimizing for latency (<italic>latencyOptim</italic>) and throughput (<italic>throughputOptim</italic>).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1268317-g0007.tif"/>
</fig>
<fig id="F8" position="float">
<label>Figure 8</label>
<caption><p>Quantization policies found by LRMP for ResNet18 while optimizing for latency and throughput.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1268317-g0008.tif"/>
</fig>
<p>In the baseline case, we observe that the latency of the network is bottlenecked by the first layer, which happens to consume very few tiles. When the layers are replicated for latency optimization (<italic>latencyOptim</italic>), the total latency is reduced by a factor of 4.6&#x000D7;, while the latency of the bottleneck layer is reduced by 14&#x000D7; as 13 more copies of that layer are created. In the throughput optimization mode (<italic>throughputOptim</italic>), the total latency is reduced by a slightly smaller factor of 4.4&#x000D7;, while the latency of the bottleneck layer is reduced by a larger factor of 19&#x000D7; as 18 more copies of that layer are created. This is understandable, because the bottleneck layer is solely responsible for determining throughput, while all layers contribute to latency. It can be observed that LRMP significantly improves tile utlization by balancing the pipeline stages through quantization and replication, resulting in an energy efficiency of 820 GOPS/s/W with <italic>throughputOptim</italic>, improving from 127 GOPS/s/W.</p>
</sec>
<sec>
<title>6.5 Analysis of sensitivity to chip area</title>
<p>As discussed in Section 5, the layer replication methodology is performed with an area constraint based on the fixed precision baseline i.e., <italic>N</italic><sub><italic>tiles</italic></sub> in the optimization constraints is equal to the number of tiles required by the fixed-precision 8-bit baseline network <italic>baseline</italic>_<italic>tiles</italic>. We note that a different design choice, based on the chip area and power budgets, could result in the relaxation or tightening of this tiles constraint.</p>
<p><xref ref-type="fig" rid="F9">Figure 9</xref> shows the sensitivity of the latency improvements achieved by LRMP to different area constraints for the ResNet18 DNN. We perform this analysis by setting <italic>N</italic><sub><italic>tiles</italic></sub> to different ratios of <italic>baseline</italic>_<italic>tiles</italic> and using LRMP to perform only quantization, only replication, and joint quantization and replication. In other words, we study the behavior of LRMP by tightening the tiles constraint below the number of tiles required by the baseline or by relaxing the tiles constraint by making more tiles available in the system, while also using only one of the two optimization dimensions of LRMP.</p>
<fig id="F9" position="float">
<label>Figure 9</label>
<caption><p>Latency improvements achieved by the proposed approach on ResNet18 with different area constraints.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1268317-g0009.tif"/>
</fig>
<p>Because of the model compression naturally achieved by mixed precision, with <italic>only</italic> mixed precision, we achieve 15.75% reduction in latency while using 39% fewer tiles than the baseline. When we employ mixed precision <italic>and</italic> layer replication, we observe latency reductions of 48% while using 35% fewer tiles than the baseline.</p>
<p>We note that layer replication can be performed even without mixed precision, if more tiles are available. We employ <italic>only</italic> layer replication with the baseline ResNet18 and observe 32% reduction in latency while using 5% more tiles than the baseline. It should be noted that when the tiles constraint is tightened, latency reductions are not possible without mixed precision, as there are not enough tiles for even a single copy of all the layers. Also, when all the tiles in the system are used, using mixed precision <italic>and</italic> layer replication achieves 46% lower latency compared to using only replication.</p>
</sec>
</sec>
<sec id="s7">
<title>7 Related works</title>
<p>In this section, we discuss related previous work in the areas of quantization and pruning for DNN implementation on IMC hardware, as well as optimized mapping of DNNs to IMC substrates.</p>
<sec>
<title>7.1 Quantization</title>
<p>Quantization is the process of reducing the number of bits used to represent numbers, which naturally adds distortions in the form of quantization noise. Quantizing weights and activations of neural networks is a common technique to reduce the model size and improve performance and energy efficiency. Pruning is a technique that removes connections and neurons in a neural network to improve sparsity. These complementary techniques have been widely explored to optimize neural network implementations. Quantization and pruning techniques have also been applied specifically to the context of in-memory computing. Peng et al. (<xref ref-type="bibr" rid="B31">2022</xref>) proposed a neural architecture search-based approach to perform mixed precision quantization in a crossbar-aware manner. Kang et al. (<xref ref-type="bibr" rid="B19">2021</xref>) proposed a methodology to perform energy-aware quantization using a genetic algorithm. Huang et al. (<xref ref-type="bibr" rid="B12">2021</xref>) proposed a methodology that performs quantization at the tile granularity powered by reinforcement learning. Meng et al. (<xref ref-type="bibr" rid="B29">2021b</xref>) proposed a quantization and pruning framework for efficient RRAM IMC implementations.</p>
</sec>
<sec>
<title>7.2 Mapping optimization</title>
<p>Mapping neural networks to IMC architectures is a complex problem. Li et al. (<xref ref-type="bibr" rid="B22">2020</xref>) proposed an approach to optimize the mapping of multimodal neural networks to IMC hardware to improve throughput. Gopalakrishnan et al. (<xref ref-type="bibr" rid="B10">2020</xref>) developed a methodology to design convolutional neural networks that would map better to crossbar architectures. Peng et al. (<xref ref-type="bibr" rid="B32">2019</xref>) proposed a weight mapping methodology that would improve data reuse of convolutional layers on crossbars. He et al. (<xref ref-type="bibr" rid="B11">2022</xref>) proposed a methodology to replicate layers in a neural network based on area freed-up by optimization of peripheral circuitry.</p>
</sec>
<sec>
<title>7.3 Optimization of peripheral circuitry</title>
<p>The peripheral circuits of crossbar tiles i.e., the DAC and ADC systems are crucial parts of IMC designs. ADCs contribute to a large portion of the power and area budgets, and are thus, a major bottleneck in the design of IMC systems. Various optimized IMC designs have been proposed that address these bottlenecks. Jiang et al. (<xref ref-type="bibr" rid="B17">2021</xref>) discusses an ADC design that implements shifts and adds in the analog domain. Saxena et al. (<xref ref-type="bibr" rid="B36">2022</xref>) proposed replacing ADCs with 1-bit sense amplifiers and training neural networks to be tolerant to such aggressive partial sum quantization. He et al. (<xref ref-type="bibr" rid="B11">2022</xref>) proposed an approach of decreasing the row parallelism to reduce the area overhead of ADCs and thus, improve the effective density of IMC chips.</p>
<p>To the best of our knowledge, LRMP is the first work that proposes a synergistic methodology that combine the benefits of mixed precision quantization and mapping optimization to jointly optimize the performance and accuracy of IMC-based neural network accelerators. LRMP is also the first work that proposes a linear programming-based approach to perform layer replication in IMC systems. Furthermore, circuit optimizations of tile peripheral are largely complementary to LRMP, and can be used to further improve the performance.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="s8">
<title>8 Conclusion</title>
<p>In-memory computing is a promising technology for accelerating neural networks by performing vector matrix multiplications within memory arrays. We propose LRMP, a method to synergistically perform layer replication and mixed precision quantization to improve performance of DNNs when mapped to area-constrained IMC accelerators. Our experiments suggest that LRMP can achieve considerable improvements in latency, throughput and energy consumption with similar accuracy compared to 8-bit fixed point implementations.</p>
</sec>
<sec sec-type="data-availability" id="s9">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://www.image-net.org/">https://www.image-net.org/</ext-link>, <ext-link ext-link-type="uri" xlink:href="http://yann.lecun.com/exdb/mnist/">http://yann.lecun.com/exdb/mnist/</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="s10">
<title>Author contributions</title>
<p>AN: Conceptualization, Methodology, Software, Validation, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. CB: Conceptualization, Writing &#x02013; review &#x00026; editing. WH: Methodology, Validation, Writing &#x02013; review &#x00026; editing. AR: Conceptualization, Funding acquisition, Methodology, Project administration, Resources, Supervision, Validation, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing.</p>
</sec>
</body>
<back>
<sec sec-type="funding-information" id="s11">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was supported in part by the U.S. Department of Energy, Office of Science, for support of microelectronics research, under contract number DE-AC0206CH11357 and in part by the National Science Foundation under grant CCF-2106964.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ankit</surname> <given-names>A.</given-names></name> <name><surname>Hajj</surname> <given-names>I. E.</given-names></name> <name><surname>Chalamalasetti</surname> <given-names>S. R.</given-names></name> <name><surname>Ndu</surname> <given-names>G.</given-names></name> <name><surname>Foltin</surname> <given-names>M.</given-names></name> <name><surname>Williams</surname> <given-names>R. S.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>&#x0201C;Puma: A programmable ultra-efficient memristor-based accelerator for machine learning inference,&#x0201D;</article-title> in <source>Proceedings of the Twenty-Fourth International Conference on Architectural Support for Programming Languages and Operating Systems</source>, <fpage>715</fpage>&#x02013;<lpage>731</lpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Asghari</surname> <given-names>M.</given-names></name> <name><surname>Fathollahi-Fard</surname> <given-names>A. M.</given-names></name> <name><surname>Mirzapour Al-e hashem</surname> <given-names>S.</given-names></name> <name><surname>Dulebenets</surname> <given-names>M. A.</given-names></name></person-group> (<year>2022</year>). <article-title>Transformation and linearization techniques in optimization: a state-of-the-art survey</article-title>. <source>Mathematics</source> <volume>10</volume>:<fpage>283</fpage>. <pub-id pub-id-type="doi">10.3390/math10020283</pub-id><pub-id pub-id-type="pmid">36081834</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Burr</surname> <given-names>G. W.</given-names></name> <name><surname>Shelby</surname> <given-names>R. M.</given-names></name> <name><surname>Sidler</surname> <given-names>S.</given-names></name> <name><surname>Di Nolfo</surname> <given-names>C.</given-names></name> <name><surname>Jang</surname> <given-names>J.</given-names></name> <name><surname>Boybat</surname> <given-names>I.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Experimental demonstration and tolerancing of a large-scale neural network (165 000 synapses) using phase-change memory as the synaptic weight element</article-title>. <source>IEEE Trans. Electron Devices</source> <volume>62</volume>, <fpage>3498</fpage>&#x02013;<lpage>3507</lpage>. <pub-id pub-id-type="doi">10.1109/TED.2015.2439635</pub-id></citation>
</ref>
<ref id="B4">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chang</surname> <given-names>M.</given-names></name> <name><surname>Spetalnick</surname> <given-names>S. D.</given-names></name> <name><surname>Crafton</surname> <given-names>B.</given-names></name> <name><surname>Khwa</surname> <given-names>W.-S.</given-names></name> <name><surname>Chih</surname> <given-names>Y.-D.</given-names></name> <name><surname>Chang</surname> <given-names>M.-F.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>&#x0201C;A 40nm 60.64tops/w ecc-capable compute-in-memory/digital 2.25mb/768kb rram/sram system with embedded cortex m3 microprocessor for edge recommendation systems,&#x0201D;</article-title> in <source>2022 IEEE International Solid- State Circuits Conference (ISSCC)</source> (<publisher-loc>San Francisco, CA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>3</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Charan</surname> <given-names>G.</given-names></name> <name><surname>Mohanty</surname> <given-names>A.</given-names></name> <name><surname>Du</surname> <given-names>X.</given-names></name> <name><surname>Krishnan</surname> <given-names>G.</given-names></name> <name><surname>Joshi</surname> <given-names>R. V.</given-names></name> <name><surname>Cao</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>Accurate inference with inaccurate rram devices: A joint algorithm-design solution</article-title>. <source>IEEE J. Explorat. Solid-State Comp. Dev. Circ</source>. <volume>6</volume>, <fpage>27</fpage>&#x02013;<lpage>35</lpage>. <pub-id pub-id-type="doi">10.1109/JXCDC.2020.2987605</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>T.</given-names></name> <name><surname>Xu</surname> <given-names>Z.</given-names></name> <name><surname>Sun</surname> <given-names>N.</given-names></name> <name><surname>Temam</surname> <given-names>O.</given-names></name></person-group> (<year>2016</year>). <article-title>Diannao family: energy-efficient hardware accelerators for machine learning</article-title>. <source>Commun. ACM</source> <volume>59</volume>, <fpage>105</fpage>&#x02013;<lpage>112</lpage>. <pub-id pub-id-type="doi">10.1145/2996864</pub-id></citation>
</ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chi</surname> <given-names>P.</given-names></name> <name><surname>Li</surname> <given-names>S.</given-names></name> <name><surname>Xu</surname> <given-names>C.</given-names></name> <name><surname>Zhang</surname> <given-names>T.</given-names></name> <name><surname>Zhao</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Prime: A novel processing-in-memory architecture for neural network computation in reram-based main memory</article-title>. <source>ACM SIGARCH Comp. Arch. News</source> <volume>44</volume>, <fpage>27</fpage>&#x02013;<lpage>39</lpage>. <pub-id pub-id-type="doi">10.1145/3007787.3001140</pub-id></citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dettmers</surname> <given-names>T.</given-names></name> <name><surname>Lewis</surname> <given-names>M.</given-names></name> <name><surname>Belkada</surname> <given-names>Y.</given-names></name> <name><surname>Zettlemoyer</surname> <given-names>L.</given-names></name></person-group> (<year>2022</year>). <article-title>Gpt3. int8 (): 8-bit matrix multiplication for transformers at scale</article-title>. <source>Adv. Neural Inf. Process. Syst</source>. <volume>35</volume>, <fpage>30318</fpage>&#x02013;<lpage>30332</lpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>F.</given-names></name> <name><surname>Tziantzioulis</surname> <given-names>G.</given-names></name> <name><surname>Wentzlaff</surname> <given-names>D.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Computedram: In-memory compute using off-the-shelf drams,&#x0201D;</article-title> in <source>Proceedings of the 52nd Annual IEEE/ACM International Symposium on Microarchitecture</source>, <fpage>100</fpage>&#x02013;<lpage>113</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gopalakrishnan</surname> <given-names>R.</given-names></name> <name><surname>Chua</surname> <given-names>Y.</given-names></name> <name><surname>Sun</surname> <given-names>P.</given-names></name> <name><surname>Kumar</surname> <given-names>A. J. S.</given-names></name> <name><surname>Basu</surname> <given-names>A.</given-names></name></person-group> (<year>2020</year>). <article-title>Hfnet: A CNN architecture co-designed for neuromorphic hardware with a crossbar array of synapses</article-title>. <source>Front. Neurosci</source>. <volume>14</volume>:<fpage>907</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2020.00907</pub-id><pub-id pub-id-type="pmid">33192236</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Chakraborty</surname> <given-names>I.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name> <name><surname>Roy</surname> <given-names>K.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Design space and memory technology co-exploration for in-memory computing based machine learning accelerators,&#x0201D;</article-title> in <source>Proceedings of the 41st IEEE/ACM International Conference on Computer-Aided Design, ICCAD &#x00027;22</source>. <publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>.</citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>S.</given-names></name> <name><surname>Ankit</surname> <given-names>A.</given-names></name> <name><surname>Silveira</surname> <given-names>P.</given-names></name> <name><surname>Antunes</surname> <given-names>R.</given-names></name> <name><surname>Chalamalasetti</surname> <given-names>S. R.</given-names></name> <name><surname>Hajj</surname> <given-names>I. E.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>&#x0201C;Mixed precision quantization for reram-based dnn inference accelerators,&#x0201D;</article-title> in <source>2021 26th Asia and South Pacific Design Automation Conference (ASP-DAC</source>).</citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jain</surname> <given-names>S.</given-names></name> <name><surname>Raghunathan</surname> <given-names>A.</given-names></name></person-group> (<year>2019</year>). <article-title>Cxdnn: Hardware-software compensation methods for deep neural networks on resistive crossbar systems</article-title>. <source><italic>ACM Trans. Embed. Comput. Syst</italic>.</source> <volume>18</volume>, <fpage>1</fpage>&#x02013;<lpage>23</lpage>.</citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jain</surname> <given-names>S.</given-names></name> <name><surname>Ranjan</surname> <given-names>A.</given-names></name> <name><surname>Roy</surname> <given-names>K.</given-names></name> <name><surname>Raghunathan</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). <article-title>Computing in memory with spin-transfer torque magnetic ram</article-title>. <source>IEEE Trans. Very Large Scale Integrat</source>. (<italic>VLSI) Syst</italic>. <volume>26</volume>, <fpage>470</fpage>&#x02013;<lpage>483</lpage>. <pub-id pub-id-type="doi">10.1109/TVLSI.2017.2776954</pub-id><pub-id pub-id-type="pmid">25278820</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jain</surname> <given-names>S.</given-names></name> <name><surname>Sengupta</surname> <given-names>A.</given-names></name> <name><surname>Roy</surname> <given-names>K.</given-names></name> <name><surname>Raghunathan</surname> <given-names>A.</given-names></name></person-group> (<year>2020</year>). <article-title>Rxnn: A framework for evaluating deep neural networks on resistive crossbars</article-title>. <source>IEEE Trans. Comp.-Aided Desig. Integrat. Circ. Syst</source>. <volume>40</volume>, <fpage>326</fpage>&#x02013;<lpage>338</lpage>. <pub-id pub-id-type="doi">10.1109/TCAD.2020.3000185</pub-id></citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jain</surname> <given-names>S.</given-names></name> <name><surname>Tsai</surname> <given-names>H.</given-names></name> <name><surname>Chen</surname> <given-names>C.-T.</given-names></name> <name><surname>Muralidhar</surname> <given-names>R.</given-names></name> <name><surname>Boybat</surname> <given-names>I.</given-names></name> <name><surname>Frank</surname> <given-names>M. M.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>A heterogeneous and programmable compute-in-memory accelerator architecture for analog-ai using dense 2-d mesh</article-title>. <source>IEEE Trans. Very Large Scale Integrat</source>. (<italic>VLSI) Syst</italic>. <volume>31</volume>, <fpage>114</fpage>&#x02013;<lpage>127</lpage>. <pub-id pub-id-type="doi">10.1109/TVLSI.2022.3221390</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jiang</surname> <given-names>H.</given-names></name> <name><surname>Li</surname> <given-names>W.</given-names></name> <name><surname>Huang</surname> <given-names>S.</given-names></name> <name><surname>Cosemans</surname> <given-names>S.</given-names></name> <name><surname>Cosemans</surname> <given-names>S.</given-names></name> <name><surname>Cosemans</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Analog-to-digital converter design exploration for compute-in-memory accelerators</article-title>. <source>IEEE Design Test of Comp</source>. <volume>38</volume>, <fpage>1</fpage>&#x02013;<lpage>8</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jouppi</surname> <given-names>N.</given-names></name> <name><surname>Young</surname> <given-names>C.</given-names></name> <name><surname>Patil</surname> <given-names>N.</given-names></name> <name><surname>Patterson</surname> <given-names>D.</given-names></name></person-group> (<year>2018</year>). <article-title>Motivation for and evaluation of the first tensor processing unit</article-title>. <source>IEEE Micro</source> <volume>38</volume>, <fpage>10</fpage>&#x02013;<lpage>19</lpage>. <pub-id pub-id-type="doi">10.1109/MM.2018.032271057</pub-id><pub-id pub-id-type="pmid">38665141</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kang</surname> <given-names>B.</given-names></name> <name><surname>Lu</surname> <given-names>A.</given-names></name> <name><surname>Long</surname> <given-names>Y.</given-names></name> <name><surname>Kim</surname> <given-names>D. H.</given-names></name> <name><surname>Yu</surname> <given-names>S.</given-names></name> <name><surname>Yu</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Genetic algorithm based energy-aware CNN quantization for processing-in-memory architecture</article-title>. <source>IEEE J. Emerg. Select. Topics Circ. Syst</source>. <volume>11</volume>, <fpage>649</fpage>&#x02013;<lpage>662</lpage>. <pub-id pub-id-type="doi">10.1109/JETCAS.2021.3127129</pub-id></citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kang</surname> <given-names>M.</given-names></name> <name><surname>Gonugondla</surname> <given-names>S. K.</given-names></name> <name><surname>Shanbhag</surname> <given-names>N. R.</given-names></name></person-group> (<year>2020</year>). <article-title>Deep in-memory architectures in sram: an analog approach to approximate computing</article-title>. <source>Proc. IEEE</source> <volume>108</volume>, <fpage>2251</fpage>&#x02013;<lpage>2275</lpage>. <pub-id pub-id-type="doi">10.1109/JPROC.2020.3034117</pub-id></citation>
</ref>
<ref id="B21">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Khaddam-Aljameh</surname> <given-names>R.</given-names></name> <name><surname>Stanisavljevic</surname> <given-names>M.</given-names></name> <name><surname>Mas</surname> <given-names>J. F.</given-names></name> <name><surname>Karunaratne</surname> <given-names>G.</given-names></name> <name><surname>Braendli</surname> <given-names>M.</given-names></name> <name><surname>Liu</surname> <given-names>F.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>&#x0201C;Hermes core-a 14nm cmos and pcm-based in-memory compute core using an array of 300ps/lsb linearized cco-based adcs and local digital processing,&#x0201D;</article-title> in <source>2021 Symposium on VLSI Circuits</source> (<publisher-loc>Kyoto</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>2</lpage>.</citation>
</ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>B.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;HITM,&#x0201D;</article-title> in <source>Proceedings of the 39th International Conference on Computer-Aided Design</source>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>W.</given-names></name> <name><surname>Han</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name></person-group> (<year>2023</year>). <article-title>Mathematical framework for optimizing crossbar allocation for reram-based CNN accelerators</article-title>. <source>ACM Trans. Des. Autom. Electron. Syst</source>. <volume>29</volume>:<fpage>1</fpage>. <pub-id pub-id-type="doi">10.1145/3631523</pub-id></citation>
</ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Lian</surname> <given-names>L.</given-names></name> <name><surname>Yang</surname> <given-names>H.</given-names></name> <name><surname>Dong</surname> <given-names>Z.</given-names></name> <name><surname>Kang</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>&#x0201C;Q-diffusion: Quantizing diffusion models,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF International Conference on Computer Vision</source>, <fpage>17535</fpage>&#x02013;<lpage>17545</lpage>.</citation>
</ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liang</surname> <given-names>T.</given-names></name> <name><surname>Glossner</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>L.</given-names></name> <name><surname>Shi</surname> <given-names>S.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name></person-group> (<year>2021</year>). <article-title>Pruning and quantization for deep neural network acceleration: a survey</article-title>. <source>Neurocomputing</source> <volume>461</volume>, <fpage>370</fpage>&#x02013;<lpage>403</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2021.07.045</pub-id></citation>
</ref>
<ref id="B26">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Lie</surname> <given-names>S.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Cerebras architecture deep dive: first look inside the hw/sw co-design for deep learning: Cerebras systems,&#x0201D;</article-title> in <source>2022 IEEE Hot Chips 34 Symposium (HCS)</source> (<publisher-loc>Cupertino, CA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>34</lpage>.</citation>
</ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lu</surname> <given-names>A.</given-names></name> <name><surname>Peng</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>W.</given-names></name> <name><surname>Jiang</surname> <given-names>H.</given-names></name> <name><surname>Yu</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>Neurosim simulator for compute-in-memory hardware accelerator: validation and benchmark</article-title>. <source>Front. Artif. Intellig</source>. <volume>4</volume>:<fpage>659060</fpage>. <pub-id pub-id-type="doi">10.3389/frai.2021.659060</pub-id><pub-id pub-id-type="pmid">34179768</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Meng</surname> <given-names>J.</given-names></name> <name><surname>Shim</surname> <given-names>W.</given-names></name> <name><surname>Yang</surname> <given-names>L.</given-names></name> <name><surname>Yeo</surname> <given-names>I.</given-names></name> <name><surname>Fan</surname> <given-names>D.</given-names></name> <name><surname>Yu</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2021a</year>). <article-title>Temperature-resilient rram-based in-memory computing for DNN inference</article-title>. <source>IEEE Micro</source> <volume>42</volume>, <fpage>89</fpage>&#x02013;<lpage>98</lpage>. <pub-id pub-id-type="doi">10.1109/MM.2021.3131114</pub-id></citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Meng</surname> <given-names>J.</given-names></name> <name><surname>Yang</surname> <given-names>L.</given-names></name> <name><surname>Peng</surname> <given-names>X.</given-names></name> <name><surname>Yu</surname> <given-names>S.</given-names></name> <name><surname>Fan</surname> <given-names>D.</given-names></name> <name><surname>sun Seo</surname> <given-names>J.</given-names></name></person-group> (<year>2021b</year>). <article-title>Structured pruning of rram crossbars for efficient in-memory computing acceleration of deep neural networks</article-title>. <source>IEEE Trans. Circ. Syst. II-Express Briefs</source>. <volume>68</volume>, <fpage>1576</fpage>&#x02013;<lpage>1580</lpage>. <pub-id pub-id-type="doi">10.1109/TCSII.2021.3069011</pub-id></citation>
</ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Narayanan</surname> <given-names>P.</given-names></name> <name><surname>Ambrogio</surname> <given-names>S.</given-names></name> <name><surname>Okazaki</surname> <given-names>A.</given-names></name> <name><surname>Hosokawa</surname> <given-names>K.</given-names></name> <name><surname>Tsai</surname> <given-names>H.</given-names></name> <name><surname>Nomura</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Fully on-chip mac at 14 nm enabled by accurate row-wise programming of PCM-based weights and parallel vector-transport in duration-format</article-title>. <source>IEEE Trans. Electron Devices</source> <volume>68</volume>, <fpage>6629</fpage>&#x02013;<lpage>6636</lpage>. <pub-id pub-id-type="doi">10.1109/TED.2021.3115993</pub-id></citation>
</ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Peng</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name> <name><surname>Zhao</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>Z.</given-names></name> <name><surname>Liu</surname> <given-names>S.</given-names></name> <name><surname>Li</surname> <given-names>Q.</given-names></name></person-group> (<year>2022</year>). <article-title>CMQ: Crossbar-aware neural network mixed-precision quantization via differentiable architecture search</article-title>. <source>IEEE Trans. Comput.-Aided Des. Integr</source>. <volume>41</volume>, <fpage>4124</fpage>&#x02013;<lpage>4133</lpage>. <pub-id pub-id-type="doi">10.1109/TCAD.2022.3197495</pub-id></citation>
</ref>
<ref id="B32">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Peng</surname> <given-names>X.</given-names></name> <name><surname>Liu</surname> <given-names>R.</given-names></name> <name><surname>Yu</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Optimizing weight mapping and data flow for convolutional neural networks on rram based processing-in-memory architecture,&#x0201D;</article-title> in <source>2019 IEEE International Symposium on Circuits and Systems (ISCAS)</source> (<publisher-loc>Sapporo</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>5</lpage>.</citation>
</ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Perez</surname> <given-names>E.</given-names></name> <name><surname>Mahadevaiah</surname> <given-names>M. K.</given-names></name> <name><surname>Quesada</surname> <given-names>E. P.-B.</given-names></name> <name><surname>Wenger</surname> <given-names>C.</given-names></name></person-group> (<year>2021</year>). <article-title>Variability and energy consumption tradeoffs in multilevel programming of rram arrays</article-title>. <source>IEEE Trans. Electron Devices</source> <volume>68</volume>, <fpage>2693</fpage>&#x02013;<lpage>2698</lpage>. <pub-id pub-id-type="doi">10.1109/TED.2021.3072868</pub-id></citation>
</ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rasch</surname> <given-names>M. J.</given-names></name> <name><surname>Gokmen</surname> <given-names>T.</given-names></name> <name><surname>Rigotti</surname> <given-names>M.</given-names></name> <name><surname>Haensch</surname> <given-names>W.</given-names></name></person-group> (<year>2019</year>). <article-title>Rapa-convnets: Modified convolutional networks for accelerated training on architectures with analog arrays</article-title>. <source>Front. Neurosci</source>. <volume>13</volume>:<fpage>753</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2019.00753</pub-id><pub-id pub-id-type="pmid">31417340</pub-id></citation></ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Roy</surname> <given-names>S.</given-names></name> <name><surname>Sridharan</surname> <given-names>S.</given-names></name> <name><surname>Jain</surname> <given-names>S.</given-names></name> <name><surname>Raghunathan</surname> <given-names>A.</given-names></name></person-group> (<year>2021</year>). <article-title>Txsim: modeling training of deep neural networks on resistive crossbar systems</article-title>. <source>IEEE Trans. Very Large Scale Integrat</source>. (<italic>VLSI) Syst</italic>. <volume>29</volume>, <fpage>730</fpage>&#x02013;<lpage>738</lpage>. <pub-id pub-id-type="doi">10.1109/TVLSI.2021.3063543</pub-id></citation>
</ref>
<ref id="B36">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Saxena</surname> <given-names>U.</given-names></name> <name><surname>Chakraborty</surname> <given-names>I.</given-names></name> <name><surname>Roy</surname> <given-names>K.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Towards ADC-less compute-in-memory accelerators for energy efficient deep learning,&#x0201D;</article-title> in <source>Design, Automation and Test in Europe</source> (<publisher-loc>Antwerp</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation>
</ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shafiee</surname> <given-names>A.</given-names></name> <name><surname>Nag</surname> <given-names>A.</given-names></name> <name><surname>Muralimanohar</surname> <given-names>N.</given-names></name> <name><surname>Balasubramonian</surname> <given-names>R.</given-names></name> <name><surname>Strachan</surname> <given-names>J. P.</given-names></name> <name><surname>Hu</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Isaac: a convolutional neural network accelerator with in-situ analog arithmetic in crossbars</article-title>. <source>ACM SIGARCH Comp. Arch. News</source> <volume>44</volume>, <fpage>14</fpage>&#x02013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1145/3007787.3001139</pub-id></citation>
</ref>
<ref id="B38">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Shim</surname> <given-names>W.</given-names></name> <name><surname>Meng</surname> <given-names>J.</given-names></name> <name><surname>Peng</surname> <given-names>X.</given-names></name> <name><surname>Seo</surname> <given-names>J.</given-names></name> <name><surname>Yu</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Impact of multilevel retention characteristics on rram based DNN inference engine,&#x0201D;</article-title> in <source>2021 IEEE International Reliability Physics Symposium (IRPS)</source> (<publisher-loc>Monterey, CA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>4</lpage>.</citation>
</ref>
<ref id="B39">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Song</surname> <given-names>L.</given-names></name> <name><surname>Qian</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Pipelayer: A pipelined reram-based accelerator for deep learning,&#x0201D;</article-title> in <source>2017 IEEE International Symposium on High Performance Computer Architecture (HPCA)</source> (<publisher-loc>Austin, TX</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>541</fpage>&#x02013;<lpage>552</lpage>.</citation>
</ref>
<ref id="B40">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>K.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Lin</surname> <given-names>Y.</given-names></name> <name><surname>Lin</surname> <given-names>J.</given-names></name> <name><surname>Han</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;HAQ: Hardware-aware automated quantization with mixed precision,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Long Beach, CA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>8612</fpage>&#x02013;<lpage>8620</lpage>.</citation>
</ref>
<ref id="B41">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Yan</surname> <given-names>H.</given-names></name> <name><surname>Cherian</surname> <given-names>H. R.</given-names></name> <name><surname>Ahn</surname> <given-names>E. C.</given-names></name> <name><surname>Duan</surname> <given-names>L.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Celia: a device and architecture co-design framework for stt-mram-based deep learning acceleration,&#x0201D;</article-title> in <source>Proceedings of the 2018 International Conference on Supercomputing</source> (<publisher-loc>Long Beach, CA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>149</fpage>&#x02013;<lpage>159</lpage>.</citation>
</ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yin</surname> <given-names>S.</given-names></name> <name><surname>Jiang</surname> <given-names>Z.</given-names></name> <name><surname>Seo</surname> <given-names>J.-S.</given-names></name> <name><surname>Seok</surname> <given-names>M.</given-names></name></person-group> (<year>2020</year>). <article-title>Xnor-sram: In-memory computing sram macro for binary/ternary deep neural networks</article-title>. <source>IEEE J. Solid-State Circuits</source> <volume>55</volume>, <fpage>1733</fpage>&#x02013;<lpage>1743</lpage>. <pub-id pub-id-type="doi">10.1109/JSSC.2019.2963616</pub-id></citation>
</ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Verma</surname> <given-names>N.</given-names></name></person-group> (<year>2017</year>). <article-title>In-memory computation of a machine-learning classifier in a standard 6t sram array</article-title>. <source>IEEE J. Solid-State Circuits</source> <volume>52</volume>, <fpage>915</fpage>&#x02013;<lpage>924</lpage>. <pub-id pub-id-type="doi">10.1109/JSSC.2016.2642198</pub-id></citation>
</ref>
<ref id="B44">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>Z.</given-names></name> <name><surname>Sun</surname> <given-names>H.</given-names></name> <name><surname>Lin</surname> <given-names>Y.</given-names></name> <name><surname>Dai</surname> <given-names>G.</given-names></name> <name><surname>Xia</surname> <given-names>L.</given-names></name> <name><surname>Han</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>&#x0201C;A configurable multi-precision CNN computing framework based on single bit RRAM,&#x0201D;</article-title> in <source>Proceedings of the 56th Annual Design Automation Conference 2019</source> (<publisher-loc>Las Vegas</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>6</lpage>.</citation>
</ref>
</ref-list>
</back>
</article>