<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2024.1371988</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Hyperdimensional computing with holographic and adaptive encoder</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Hern&#x000E1;ndez-Cano</surname> <given-names>Alejandro</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Ni</surname> <given-names>Yang</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2631380/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zou</surname> <given-names>Zhuowen</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1643701/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zakeri</surname> <given-names>Ali</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Imani</surname> <given-names>Mohsen</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1392248/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Computer Science, &#x000C9;cole polytechnique f&#x000E9;d&#x000E9;rale de Lausanne (EPFL)</institution>, <addr-line>Lausanne</addr-line>, <country>Switzerland</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Computer Science, University of California, Irvine</institution>, <addr-line>Irvine, CA</addr-line>, <country>United States</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Giner Alor-Hern&#x000E1;ndez, Instituto Tecnologico de Orizaba, Mexico</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Evgeny Osipov, Lule&#x000E5; University of Technology, Sweden</p>
<p>Nelson Rangel-Valdez, Instituto Tecnol&#x000F3;gico de Ciudad Madero, Mexico</p>
<p>Gilberto Rivera, Universidad Aut&#x000F3;noma de Ciudad Ju&#x000E1;rez, Mexico</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Mohsen Imani <email>m.imani&#x00040;uci.edu</email></corresp></author-notes>
<pub-date pub-type="epub">
<day>09</day>
<month>04</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>7</volume>
<elocation-id>1371988</elocation-id>
<history>
<date date-type="received">
<day>17</day>
<month>01</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>18</day>
<month>03</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2024 Hern&#x000E1;ndez-Cano, Ni, Zou, Zakeri and Imani.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Hern&#x000E1;ndez-Cano, Ni, Zou, Zakeri and Imani</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Brain-inspired computing has become an emerging field, where a growing number of works focus on developing algorithms that bring machine learning closer to human brains at the functional level. As one of the promising directions, Hyperdimensional Computing (HDC) is centered around the idea of having holographic and high-dimensional representation as the neural activities in our brains. Such representation is the fundamental enabler for the efficiency and robustness of HDC. However, existing HDC-based algorithms suffer from limitations within the encoder. To some extent, they all rely on manually selected encoders, meaning that the resulting representation is never adapted to the tasks at hand.</p></sec>
<sec>
<title>Methods</title>
<p>In this paper, we propose FLASH, a novel hyperdimensional learning method that incorporates an adaptive and learnable encoder design, aiming at better overall learning performance while maintaining good properties of HDC representation. Current HDC encoders leverage Random Fourier Features (RFF) for kernel correspondence and enable locality-preserving encoding. We propose to learn the encoder matrix distribution via gradient descent and effectively adapt the kernel for a more suitable HDC encoding.</p></sec>
<sec>
<title>Results</title>
<p>Our experiments on various regression datasets show that tuning the HDC encoder can significantly boost the accuracy, surpassing the current HDC-based algorithm and providing faster inference than other baselines, including RFF-based kernel ridge regression.</p></sec>
<sec>
<title>Discussion</title>
<p>The results indicate the importance of an adaptive encoder and customized high-dimensional representation in HDC.</p></sec></abstract>
<kwd-group>
<kwd>brain-inspired computing</kwd>
<kwd>hyperdimensional computing</kwd>
<kwd>holographic representation</kwd>
<kwd>vector function architecture</kwd>
<kwd>efficient machine learning</kwd>
</kwd-group>
<contract-num rid="cn001">Young Faculty Award</contract-num>
<contract-num rid="cn002">2127780</contract-num>
<contract-num rid="cn002">2235472</contract-num>
<contract-num rid="cn002">2312517</contract-num>
<contract-num rid="cn002">2319198</contract-num>
<contract-num rid="cn002">2321840</contract-num>
<contract-num rid="cn004">N00014-21-1-2225</contract-num>
<contract-num rid="cn004">N00014-22-1-2067</contract-num>
<contract-num rid="cn004">Young Investigator Program Award</contract-num>
<contract-num rid="cn005">FA9550-22-1-0253</contract-num>
<contract-sponsor id="cn001">Defense Advanced Research Projects Agency<named-content content-type="fundref-id">10.13039/100000185</named-content></contract-sponsor>
<contract-sponsor id="cn002">National Science Foundation<named-content content-type="fundref-id">10.13039/100000001</named-content></contract-sponsor>
<contract-sponsor id="cn003">Semiconductor Research Corporation<named-content content-type="fundref-id">10.13039/100000028</named-content></contract-sponsor>
<contract-sponsor id="cn004">Office of Naval Research<named-content content-type="fundref-id">10.13039/100000006</named-content></contract-sponsor>
<contract-sponsor id="cn005">Air Force Office of Scientific Research<named-content content-type="fundref-id">10.13039/100000181</named-content></contract-sponsor>
<contract-sponsor id="cn006">Cisco Systems<named-content content-type="fundref-id">10.13039/100004351</named-content></contract-sponsor>
<counts>
<fig-count count="5"/>
<table-count count="3"/>
<equation-count count="10"/>
<ref-count count="40"/>
<page-count count="11"/>
<word-count count="7738"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Machine Learning and Artificial Intelligence</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>The human brain remains the most sophisticated yet effective learning module ever. Running on similar power of light bulbs, our brains are in charge of almost every learning and reasoning task in daily life with particularly great sample efficiency and fault tolerance. On the contrary, many widely-applied Machine Learning (ML) algorithms fail to be comparable in efficiency and robustness, despite their prolific advancement in accomplishing practical tasks.</p>
<p>Therefore, research in biological vision, cognitive psychology, and neuroscience has given rise to key concepts behind an emerging field, i.e., brain-inspired computing. In this field, several novel computing paradigms have been developed during the last few years that are either biologically plausible or closer to human brains at the functional level (Roy et al., <xref ref-type="bibr" rid="B35">2019</xref>; Karunaratne et al., <xref ref-type="bibr" rid="B16">2020</xref>). In particular, HyperDimensional Computing (HDC) mimics human brain functionalities when learning and reasoning in high-dimensional spaces (i.e., the hyperspace in HDC), which is motivated by the observation that the human brain operates on high-dimensional neural representations. Similarly in the brain-inspired HDC, a high-dimensional vector-based representation is designed to represent different atomic concepts such as letters, objects, sensor readings, and general features. Typically, an HDC encoder will encode inputs from the original lower-dimensional space to very high-dimensional vectors (i.e., hypervectors in HDC) with several thousand dimensions. Centered on the hypervectors, HDC is also capable of describing the location of objects, their relations, and the structured combination of several individual concepts through a set of HDC operations designed for hypervectors (more details in Section 2).</p>
<p>As the basic building block of HDC, hypervectors own several unique properties that have been crucial for practical applications, especially in terms of representing and manipulating atomic symbols. Specifically, hypervector representation is (1) <italic>holographic</italic>, that information is distributed evenly across components of the hypervector (Kleyko et al., <xref ref-type="bibr" rid="B18">2023</xref>), (2) <italic>robust</italic>, that hypervectors are extremely noise tolerant as a natural result of hypervector redundancy (Kanerva, <xref ref-type="bibr" rid="B15">2009</xref>; Poduval et al., <xref ref-type="bibr" rid="B30">2022b</xref>; Barkam et al., <xref ref-type="bibr" rid="B1">2023a</xref>), and (3) <italic>simple</italic>, that only lightweight operations are needed to perform learning tasks (Hernandez-Cane et al., <xref ref-type="bibr" rid="B10">2021</xref>; Ni et al., <xref ref-type="bibr" rid="B24">2022b</xref>). In addition, the ability for hypervectors to operate symbolically through simple arithmetic has granted HDC the ability to perform cognitive tasks in a transparent and compositional way, e.g., memorization, learning, and association (Poduval et al., <xref ref-type="bibr" rid="B29">2022a</xref>; Hersche et al., <xref ref-type="bibr" rid="B12">2023</xref>). Given the importance of the properties aforementioned, most HDC frameworks have a dedicated and specially designed HDC encoder for mapping original inputs to corresponding hypervectors. The quality of encoded hyperdimensional representations can be decisive for performance in learning and cognitive tasks.</p>
<p>While the HDC encoder has had many variants (Rachkovskij, <xref ref-type="bibr" rid="B31">2015</xref>; Kleyko et al., <xref ref-type="bibr" rid="B19">2018</xref>, <xref ref-type="bibr" rid="B17">2021</xref>; Imani et al., <xref ref-type="bibr" rid="B13">2019</xref>; Frady et al., <xref ref-type="bibr" rid="B7">2021</xref>), most of them innovate on the encoding scheme, i.e., the way symbolically different entities are encoded together. One common example is Position-ID encoding (Thomas et al., <xref ref-type="bibr" rid="B37">2020</xref>): each feature is assigned a (key) hypervector representing its position in the vector, and the value of the feature is quantized to a set of discrete levels and assigned the corresponding (level or value) hypervector. The representation of a feature vector is thus a bundling of several binding key-value pairs. Despite the success of mentioned encoding, the quality of HDC representation of atomic hypervectors is ambiguous: their design is barely discussed due to the already competitive richness in representation (Park and Sandberg, <xref ref-type="bibr" rid="B28">1991</xref>) and performance in practice (Ge and Parhi, <xref ref-type="bibr" rid="B9">2020</xref>). In the case of Position-ID encoding, for example, key hypervectors are assumed to be independent of each other, while value hypervectors preserve a discrete linear similarity with each other. Such manually selected similarity metrics, linear mapping, and discrete atomic compositions naturally lack flexibility. Recognizing this gap, we ask in this paper a fundamental question in improving HDC learning: how can we generate good HDC representation for atomic data? And also, how can we create an encoding scheme that adapts to the problem at hand?</p>
<p>Recent research proposes Vector Function Architecture (Kleyko et al., <xref ref-type="bibr" rid="B17">2021</xref>; Frady et al., <xref ref-type="bibr" rid="B6">2022</xref>) (VFA) that provides a general approach for better representation of continuous data and functions in the hyperspace. Its encoder, instead of presetting discrete levels of similarity for each feature in the original space, directly targets a meaningful similarity in the whole hyperspace such as the Gaussian radial basis function. To do so, VFA relies on fractional power encoding and Random Fourier Features (Rahimi and Recht, <xref ref-type="bibr" rid="B33">2007</xref>) (RFF) parameterized by a high-dimensional encoding matrix through a predefined random distribution. The resulting hyperspace then holds a high-dimensional and non-linear representation that maintains the distance relationship in much finer granularity. We notice that HDC representation quality relies heavily on the choice of hyperspace mapping and similarity metric, which are manifested directly via the distribution from which every component of the encoding matrix is sampled. Recognizing this connection, we expect that selecting a distribution well-adapted to the task will essentially enhance the quality of the HDC encoder as well as learning performance.</p>
<p>In this paper, we bring FLASH, to the best of our knowledge, the first HDC representation that is <underline>F</underline>ast, <underline>L</underline>earnable, <underline>A</underline>daptive, and <underline>S</underline>tays <underline>H</underline>olographic. FLASH leads to an innovative hyperdimensional regression algorithm featuring an optimizable HDC encoder. Unlike all the previous algorithms that limit themselves to either prefixed atomic hypervectors or static encoding mechanisms, our method (1) generates atomic hypervectors that truly adapt to the training data at hand, (2) efficiently optimizes the HDC representation for downstream tasks, and (3) maintains the major benefits of HDC, i.e., holographic representation. We take inspiration from the prior VFA work and propose a novel mechanism to enhance the representation in hyperspace by finding the optimal distribution from which the random matrix is drawn. Moreover, this approach does not require us to use explicitly the kernel function nor the probability density, nor to perform expensive Fourier transforms. This allows the encoding process of FLASH to be as efficient as the static one with the exception of a one-time overhead for optimization.</p>
<p>Our experimental results show that FLASH is about 5.5 &#x000D7; faster in inference than RFF-based ridge regression while providing comparable or better accuracy. We also test a variant called &#x0201C;Accurate FLASH&#x0201D; that is optimized for accuracy, and this approach consistently outperforms other ML baselines, including the previous state-of-the-art HDC regression algorithm (Hern&#x000E1;ndez-Cano et al., <xref ref-type="bibr" rid="B11">2021</xref>) based on VFA. At the same time, we observe a linear increase in our approach with respect to the number of samples in the dataset, making this proposal particularly well-suited for large-scale data.</p>
<p>The rest of this article is organized as follows. In the &#x0201C;HDC Background&#x0201D; section, the basics of HDC are described. And the prior arts VFA-based hyperdimensional regression algorithm is analyzed in the &#x0201C;Regression&#x0201D; section. Our proposed FLASH is formulated in the &#x0201C;Main Methods&#x0201D; section. The &#x0201C;Experimental Results&#x0201D; section presents results for experiments carried out on multiple regression datasets. Finally, the &#x0201C;Conclusion&#x0201D; section concludes this article.</p></sec>
<sec id="s2">
<title>2 Related works</title>
<p>In the past few years, prior HDC research works have applied the brain-like functionalities of HDC to diverse applications, for example, outlier detection (Wang et al., <xref ref-type="bibr" rid="B39">2022</xref>), biosignal processing (Rahimi et al., <xref ref-type="bibr" rid="B34">2020</xref>; Ni et al., <xref ref-type="bibr" rid="B25">2022c</xref>; Pale et al., <xref ref-type="bibr" rid="B27">2022</xref>), speech recognition (Hernandez-Cane et al., <xref ref-type="bibr" rid="B10">2021</xref>), and gesture recognition (Rahimi et al., <xref ref-type="bibr" rid="B32">2016</xref>). Apart from classification learning tasks, it has also been applied to genomic sequencing (Zou et al., <xref ref-type="bibr" rid="B40">2022</xref>; Barkam et al., <xref ref-type="bibr" rid="B2">2023b</xref>), nonlinear regression (Hern&#x000E1;ndez-Cano et al., <xref ref-type="bibr" rid="B11">2021</xref>; Ni et al., <xref ref-type="bibr" rid="B22">2023b</xref>), reinforcement learning (Chen et al., <xref ref-type="bibr" rid="B3">2022</xref>; Issa et al., <xref ref-type="bibr" rid="B14">2022</xref>; Ni et al., <xref ref-type="bibr" rid="B23">2022a</xref>, <xref ref-type="bibr" rid="B21">2023a</xref>), and graph reasoning (Poduval et al., <xref ref-type="bibr" rid="B29">2022a</xref>; Chen et al., <xref ref-type="bibr" rid="B4">2023</xref>). With or without hardware acceleration, these HDC algorithms bring a significant efficiency benefit to each application, facilitating online training, few-shot learning, and edge-friendly operation. However, their performance is inevitably limited by a poorly-optimized encoding process. The mapping to hyperspace is either manually devised for a specific task or directly reuses a fixed design such as VFA (Rahimi et al., <xref ref-type="bibr" rid="B34">2020</xref>; Hern&#x000E1;ndez-Cano et al., <xref ref-type="bibr" rid="B11">2021</xref>). In this paper, we focus on improving the HDC encoder design and thus learning performance by proposing a novel encoder that is optimizable and adaptive.</p></sec>
<sec id="s3">
<title>3 Regression with vector function architecture</title>
<p>In this section, we first briefly revisit the hyperdimensional encoding technique proposed in VFA. As mentioned in the introduction, the VFA encoding mounts to a well-defined and continuous mapping to hyperspace. We then discuss its usage in the current state-of-the-art HDC regression algorithm and point out the limitation of this method due to the static encoding.</p>
<sec>
<title>3.1 Hyperdimensional encoding in VFA</title>
<p>As a symbolic paradigm, many HDC algorithms operate on a set of dissimilar atomic hypervectors that are randomly generated and near-orthogonal, assuming that symbols are not related at all. However, the assumption will not always be appropriate in practical tasks. Therefore, we have seen HDC algorithms, when handling bio-signals and images, explicitly manipulate the similarity among atomic hypervectors such as maintaining a discrete set of similarity levels. However, the manual assignment of these hypervectors can be problematic, which inevitably causes information loss during quantization, not to mention that such an arbitrarily assumed similarity relationship may not be helpful for learning.</p>
<p>To explain how the VFA encoding captures the relation between data, we note that this encoding coincides with the Random Fourier Features (RFF) encoding, an efficient approximation of kernel methods. The following theorem by Salomon Bochner (1899&#x02013;1982) serves as the foundation for this well-defined similarity relationship (Rudin, <xref ref-type="bibr" rid="B36">2017</xref>).</p>
<p><bold> Theorem 1 (Bochner)</bold>. For any continuous shift-invariant and positive definite kernel <inline-formula><mml:math id="M1"><mml:mrow><mml:mi>K</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mn>1</mml:mn></mml:mstyle></mml:msub><mml:mo>&#x02212;</mml:mo><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mn>2</mml:mn></mml:mstyle></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>:</mml:mo><mml:msup><mml:mi>&#x0211D;</mml:mi><mml:mi>M</mml:mi></mml:msup><mml:mo>&#x02192;</mml:mo><mml:mi>&#x0211D;</mml:mi></mml:mrow></mml:math></inline-formula>, there must exist a non-negative measure <italic>p</italic>(<bold>&#x003C9;</bold>) such that <italic>K</italic> is the Fourier transform of a non-negative measure <italic>p</italic>(<bold>&#x003C9;</bold>). Additionally, if <italic>K</italic> is properly scaled, <italic>p</italic>(<bold>&#x003C9;</bold>) is a proper probability measure.</p>
<p>The proof of this theorem is provided in Rudin (<xref ref-type="bibr" rid="B36">2017</xref>). If we assume &#x003B6;(<bold>x</bold>) &#x0003D; <italic>e</italic><sup><italic>j</italic><bold>&#x003C9;</bold></sup><sup><italic>T</italic></sup><bold>x</bold>, then Theorem 1 leads to the following equation: <inline-formula><mml:math id="M2"><mml:mrow><mml:mi>K</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mn>1</mml:mn></mml:mstyle></mml:msub><mml:mo>&#x02212;</mml:mo><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mn>2</mml:mn></mml:mstyle></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mstyle displaystyle='true'><mml:mrow><mml:msub><mml:mo>&#x0222B;</mml:mo><mml:mrow><mml:msup><mml:mi>&#x0211D;</mml:mi><mml:mi>M</mml:mi></mml:msup></mml:mrow></mml:msub><mml:mi>p</mml:mi></mml:mrow></mml:mstyle><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003C9;</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:msup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003C9;</mml:mi></mml:mstyle><mml:mi>T</mml:mi></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mn>1</mml:mn></mml:mstyle></mml:msub><mml:mo>&#x02212;</mml:mo><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mn>2</mml:mn></mml:mstyle></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:msub><mml:mi mathvariant='double-struck'>E</mml:mi><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003C9;</mml:mi></mml:mstyle></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>&#x003B6;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mover accent='true'><mml:mrow><mml:mi>&#x003B6;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo stretchy='true'>&#x000AF;</mml:mo></mml:mover></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>. This means that, with the correspondence between kernel <italic>K</italic> and measure <italic>p</italic>(<bold>x</bold>), we can transform original inputs to a space where dot products are unbiased estimates of kernel similarities. In other words, there exist sequences of transformations <inline-formula><mml:math id="M3"><mml:msub><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:msub><mml:mo>:</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msup><mml:mo>&#x02192;</mml:mo><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>C</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> such that <inline-formula><mml:math id="M4"><mml:mrow><mml:mtext>&#x000A0;</mml:mtext><mml:msub><mml:mi>&#x003D5;</mml:mi><mml:mi>D</mml:mi></mml:msub><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mn>1</mml:mn></mml:mstyle></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mi>T</mml:mi></mml:msup><mml:msub><mml:mi>&#x003D5;</mml:mi><mml:mi>D</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mn>2</mml:mn></mml:mstyle></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:math></inline-formula> converges uniformly to the given kernel <italic>K</italic>(<bold>x<sub>1</sub></bold>&#x02212;<bold>x<sub>2</sub></bold>):</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M5"><mml:mrow><mml:msub><mml:mi>&#x003D5;</mml:mi><mml:mi>D</mml:mi></mml:msub><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mn>1</mml:mn></mml:mstyle></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mi>T</mml:mi></mml:msup><mml:mover accent='true'><mml:mrow><mml:msub><mml:mi>&#x003D5;</mml:mi><mml:mi>D</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mn>2</mml:mn></mml:mstyle></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo stretchy='true'>&#x000AF;</mml:mo></mml:mover><mml:mover><mml:mo>&#x02192;</mml:mo><mml:mrow><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>D</mml:mi></mml:mrow></mml:mover><mml:mi>K</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mn>1</mml:mn></mml:mstyle></mml:msub><mml:mo>&#x02212;</mml:mo><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mn>2</mml:mn></mml:mstyle></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:math></disp-formula>
<p>Rahimi and Recht (<xref ref-type="bibr" rid="B33">2007</xref>) proposed an alternative set of Random Fourier Features (RFF) such that the components of the encoded vectors are real and the kernel approximation converges equally fast. To construct a real-valued RFF, we can leverage this high-dimensional mapping for HDC encoder as the following:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>&#x003A9;</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msqrt><mml:mrow><mml:mfrac><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:msqrt><mml:mo class="qopname">cos.</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003A9;</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>&#x0002B;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where cos. represents the element-wise cosine function, <bold>&#x003A9;</bold>&#x02208;&#x0211D;<sup><italic>D</italic>&#x000D7;<italic>M</italic></sup> is a randomly generated encoding matrix, and <bold>b</bold>&#x02208;&#x0211D;<sup><italic>D</italic></sup> is the offset hypervector. Row vectors in <bold>&#x003A9;</bold> are generated by drawing <italic>D</italic> i.i.d. samples <bold>&#x003C9;</bold><sub>1</sub>, &#x02026;, <bold>&#x003C9;</bold><sub><italic>D</italic></sub> from <italic>p</italic>(<bold>&#x003C9;</bold>) and elements in <italic>b</italic> are sampled from <inline-formula><mml:math id="M7"><mml:mrow><mml:mi mathvariant="script">U</mml:mi></mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mi>&#x003C0;</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula>. The random distribution <italic>p</italic> is selected through Theorem 1 given a preferred kernel <italic>K</italic>(&#x00394;). In other words, the probability density is calculated with a Fourier transform: <inline-formula><mml:math id="M8"><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003C9;</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mi>&#x003C0;</mml:mi></mml:mrow></mml:mfrac><mml:munder class="msub"><mml:mrow><mml:mo>&#x0222B;</mml:mo></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:munder><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>-</mml:mo><mml:mi>i</mml:mi><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003C9;</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:mi>&#x00394;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>K</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x00394;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mi>&#x00394;</mml:mi></mml:math></inline-formula>. For example, previous HDC work (Hern&#x000E1;ndez-Cano et al., <xref ref-type="bibr" rid="B11">2021</xref>) uses Normal distribution <inline-formula><mml:math id="M9"><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> for <italic>p</italic>(<bold>&#x003C9;</bold>) since it wants to approximate the Gaussian RBF kernel.</p>
<p>There are key lessons from the theory of VFA encoding discussed above:</p>
<list list-type="order">
<list-item><p>HDC encoding as in <xref ref-type="disp-formula" rid="E2">Equation (2)</xref> incorporates a high-dimensional non-linear mapping through the cosine activation.</p></list-item>
<list-item><p>Due to RFF, it supports a meaningful similarity metric in hyperspace without quantizing individual features or generating ambiguous correlated base hypervectors.</p></list-item>
<list-item><p>By Bochner&#x00027;s theorem, there is a correspondence between kernel <italic>K</italic> and measure <italic>p</italic>(<bold>x</bold>). This implies that we can leverage the measure for the estimation of kernel similarities.</p></list-item>
</list>
<p>While the current VFA method brings many benefits, the biggest drawback of this method is that &#x003D5;<sub><italic>D</italic></sub>(<bold>x</bold>) is essentially a static mapping, which makes the encoding less adaptive. Our work aims to leverage the insight from Bochner&#x00027;s theorem to learn the kernel adaptively through its random Fourier features.</p></sec>
<sec>
<title>3.2 Regression on a static HDC encoder</title>
<p>Ideally, we expect the HDC encoder to provide a useful high-dimensional representation that helps separate the data points for classification or linearize the inherent non-linear regression tasks. Particularly in hyperdimensional regression, we are interested in finding the best hypervector <bold>w</bold>&#x02208;&#x0211D;<sup><italic>D</italic></sup> such that the linear regression after encoding <italic>y</italic>(<bold>x</bold>) &#x0003D; &#x003D5;(<bold>x</bold>)<sup><italic>T</italic></sup><bold>w</bold> are, in average across the training set, as close as possible to the true labels in terms of &#x02113;<sub>2</sub> norm. Additionally, we introduce an &#x02113;<sub>2</sub> regularization coefficient &#x003BB; to get more stable estimators. Thus the loss function is the dampened least squares, which can be expressed as</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M10"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>:</mml:mo><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mo>||</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>Z</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle><mml:mo>-</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle><mml:mo>||</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003BB;</mml:mi><mml:msup><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle><mml:mo>||</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M11"><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> are the known response variables, and <bold>Z</bold> &#x0003D; &#x003A6;(<bold>X</bold>; <bold>&#x003A9;</bold>, <bold>b</bold>)&#x02208;&#x0211D;<sup><italic>N</italic>&#x000D7;<italic>D</italic></sup> are the encoded hypervectors of the input data <bold>x</bold><sub>1</sub>, &#x02026;, <bold>x</bold><sub><italic>N</italic></sub>:</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M12"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>&#x003A6;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>&#x003A9;</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mi>&#x003D5;</mml:mi><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>&#x003A9;</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>&#x022EE;</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>&#x003D5;</mml:mi><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>&#x003A9;</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msqrt><mml:mrow><mml:mfrac><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:msqrt><mml:mo class="qopname">cos.</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003A9;</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In previous approaches for HDC regression (Hern&#x000E1;ndez-Cano et al., <xref ref-type="bibr" rid="B11">2021</xref>; Kleyko et al., <xref ref-type="bibr" rid="B17">2021</xref>), the model hypervector <bold>w</bold> is learned in an iterative fashion, where hypervectors are bundled together guided by regression errors. However, they face issues with proper hyperparameter selection to achieve the highest prediction quality. On the other hand, the loss function in <xref ref-type="disp-formula" rid="E3">Equation (3</xref>) has a known minimizer <inline-formula><mml:math id="M13"><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:math></inline-formula> which is known as the ridge estimator:</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M14"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>Z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:mstyle mathvariant="bold"><mml:mtext>Z</mml:mtext></mml:mstyle><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003BB;</mml:mi><mml:mstyle mathvariant="bold"><mml:mtext>I</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>Z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In this paper, we leverage this statistical approach to obtain better stability during learning and more direct parameter tuning.</p>
<p>As we mentioned in the previous section, the encoding in VFA, the closed solution in <xref ref-type="disp-formula" rid="E5">Equation (5)</xref> has an assumption that the regression problem on <bold>Z</bold> is linearly solvable, as the result of mapping to hyperspace. However, for an arbitrary regression task, it is very likely that the static VFA encoding (due to the fixed <italic>K</italic>(&#x00394;) and <italic>p</italic>(<bold>&#x003C9;</bold>)) becomes sub-optimal. This work looks to address this problem by presenting an adaptive HDC encoder design.</p></sec></sec>
<sec id="s4">
<title>4 Main methods</title>
<p>In this paper, we proposeFLASH, a way to learn a good encoding function &#x003D5;(<bold>x</bold>) before solving the regression task in hyperdimensional space.</p>
<p>In <xref ref-type="fig" rid="F1">Figure 1</xref>, we present an outline of our proposed FLASH, including both HDC inference and encoder learning processes. In the inference process, we start from a query data point <bold>x</bold>&#x02208;&#x0211D;<sup><italic>M</italic></sup> in the original space &#x02776;, which is then passed through the encoding module &#x02777; to obtain the encoded data point <bold>z</bold>&#x02208;&#x0211D;<sup><italic>D</italic></sup> in the hyperspace. Once we have this query hypervector, getting the prediction &#x00177; reduces to perform dot product with the regression hypervector &#x02778;. The overall inference process depicted here is similar to VFA-based regression; what distinguishes FLASH from prior algorithms is that the encoding module (<bold>&#x003A9;</bold> matrix, specifically) is obtained through a parameterized distribution <italic>p</italic><sub>&#x003B8;</sub>. During the encoder learning &#x02779;, parameters in <italic>p</italic><sub>&#x003B8;</sub> are updated given the feedback from the regression loss defined in <xref ref-type="disp-formula" rid="E3">Equation (3)</xref>. In the following sections, we will introduce how to sample from this parameterized distribution and learn its parameters.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Overview of our proposed FLASH: the area with shade represents the inference process; the rest modules are related to adapting the HDC encoder, where we learn the distribution of the encoding matrix.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1371988-g0001.tif"/>
</fig>


<sec>
<title>4.1 Generating the encoding matrix</title>
<p>Care must be taken when designing the HDC encoder, as it needs to ensure that the appealing properties of the hypervectors are sustained. To ensure the holographic representation, we require the randomized instantiation of the encoder, and thus we cannot directly perform gradient descent upon an instantiated encoding matrix, as it may destroy both: information about the data may be distributed locally and partially in the output vector, and the trained encoder have minimum randomization.</p>
<p>To circumvent this, we take inspiration from the VFA work that highlights the encoder-kernel correspondence and the importance of selecting a proper distribution <italic>p</italic>(<bold>&#x003C9;</bold>). Recall from Section 3.1 that there is a correspondence between kernel <italic>K</italic> and measure <italic>p</italic>(<bold>x</bold>). In addition, it induces families of encoders {<sub>&#x003D5;<sub><italic>D</italic></sub>}<italic>D</italic>&#x02208;&#x02115;</sub> parameterized by the samples from the distribution such that the inner product between the encoded vectors approximates the kernel. This approximation improves increasingly well with the dimension of the encoder (codomain). By learning the distribution from which we sample the encoding matrix, we will be able to construct an adaptive HDC encoder that provides a more suitable hyperdimensional representation, adding to existing appealing HDC properties.</p>
<p>In FLASH, we define a parametric family of functions <inline-formula><mml:math id="M15"><mml:mrow><mml:mi mathvariant="script">F</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003B8;</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mo>:</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003B8;</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mo>:</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msup><mml:mo>&#x02192;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>&#x003B8;</mml:mtext></mml:mstyle><mml:mo>&#x02208;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>&#x00398;</mml:mtext></mml:mstyle></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula>, which will be referred to as generators. Upon receiving random inputs, it can be used to sample random vectors <bold>&#x003C9;</bold><sub>1</sub>, &#x02026;, <bold>&#x003C9;</bold><sub><italic>D</italic></sub> required in the encoding function (<xref ref-type="disp-formula" rid="E4">Equation 4</xref>). Compared to parameterized distributions such as the Gaussian reparameterization approach, this approach encapsulates a more expressive family of distributions. Because of Bochner&#x00027;s theorem, exploring <inline-formula><mml:math id="M16"><mml:mrow><mml:mi mathvariant="script">F</mml:mi></mml:mrow></mml:math></inline-formula> is equivalent to exploring a corresponding set of continuous shift-invariant kernels, and thus the encoding family we consider is expected to be rich as long as <inline-formula><mml:math id="M17"><mml:mrow><mml:mi mathvariant="script">F</mml:mi></mml:mrow></mml:math></inline-formula> is.</p>
<p>Inherited from the RFF methods, a key benefit of this arrangement is that FLASH encoding does not require us to use explicitly the kernel <italic>K</italic>(&#x00394;) function, nor the probability density <italic>p</italic>(<bold>&#x003C9;</bold>), nor to perform expensive Fourier transforms. Our goal is to find <inline-formula><mml:math id="M18"><mml:mi>f</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mi mathvariant="script">F</mml:mi></mml:mrow></mml:math></inline-formula> that, with a high probability, gives the optimal or near-optimal encoding matrix of solving the regression problem at hand; this ensures the quality and robustness of the encoder that it generates.</p></sec>
<sec>
<title>4.2 Learning the encoder matrix distribution</title>
<p>To learn the distribution efficiently, we restrict <inline-formula><mml:math id="M19"><mml:mrow><mml:mi mathvariant="script">F</mml:mi></mml:mrow></mml:math></inline-formula> to be a family of <inline-formula><mml:math id="M20"><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003B8;</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mo>:</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msup><mml:mo>&#x02192;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> differentiable neural networks, i.e., the network input size equals the output size. To sample <bold>&#x003C9;</bold>s using <italic>f</italic><sub><bold>&#x003B8;</bold></sub>, we first draw a random vector <inline-formula><mml:math id="M21"><mml:mstyle mathvariant="bold"><mml:mtext>&#x003F5;</mml:mtext></mml:mstyle><mml:mo>&#x0007E;</mml:mo><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>0</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>I</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> as the input, and then obtain a transformed random vector using <bold>&#x003C9;</bold> &#x0003D; <italic>f</italic><sub><bold>&#x003B8;</bold></sub>(<bold>&#x003F5;</bold>). Sampling <italic>D</italic> i.i.d. random noises and passing them through the generator can give us a matrix of base hypervectors (i.e., <bold>&#x003A9;</bold>), which can be used to perform the encoding. This can be understood as a generalized reparameterization, where we learn a surrogate function <italic>f</italic><sub><bold>&#x003B8;</bold></sub>(<bold>&#x003F5;</bold>) instead. Note that we chose to sample noise vectors from the standard normal distribution for convenience, but different choices can be made as well. This architecture gives us a very rich family of generators <inline-formula><mml:math id="M22"><mml:mrow><mml:mi mathvariant="script">F</mml:mi></mml:mrow></mml:math></inline-formula>, which are cheap to evaluate and cheap to train, as we will see in the next few sections. In addition, because <bold>&#x003F5;</bold> is a random vector, <bold>&#x003C9;</bold> &#x0003D; <italic>f</italic><sub><bold>&#x003B8;</bold></sub>(<bold>&#x003F5;</bold>) is one too, which means that there exists a probability density (or mass) function <italic>p</italic><sub><bold>&#x003B8;</bold></sub>(<bold>&#x003C9;</bold>) for each generator <italic>f</italic><sub><bold>&#x003B8;</bold></sub>.</p>
<p>In FLASH, we aim to maximize encoder performance in generating a good HDC representation of our data for the regression task. In particular, we opt to evaluate the learned encoder using the loss function proposed in <xref ref-type="disp-formula" rid="E3">Equation (3)</xref>. However, because random sampling is involved at the moment of generating the encoding, we chose to minimize the expected value instead. Note that this expected value is taken with respect to all the possible encoding <bold>&#x003A6;</bold>, whose encoding matrix <bold>&#x003A9;</bold> and offset hypervectors <bold>b</bold> are randomly sampled. Thus, in <xref ref-type="disp-formula" rid="E6">Equation 6</xref>, we seek to find the parameters in <italic>f</italic><sub><bold>&#x003B8;</bold></sub> that minimize the following loss term for adapting the HDC encoder:</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M23"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003B8;</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo class="qopname">E</mml:mo></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003A9;</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow></mml:munder></mml:mstyle><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo class="qopname">min</mml:mo></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M24"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> is the regression loss defined in <xref ref-type="disp-formula" rid="E3">Equation (3)</xref>. In <xref ref-type="disp-formula" rid="E7">Equation 7</xref>, using the ridge estimator <inline-formula><mml:math id="M25"><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:math></inline-formula> and the law of the unconscious statistician, we can expand the previous expectation term to:</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M26"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:munder><mml:mtext>E</mml:mtext><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003A9;</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>b</mml:mi></mml:mstyle></mml:mrow></mml:munder><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mi>&#x02112;</mml:mi><mml:mi>R</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mover accent='true'><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='true'>&#x0005E;</mml:mo></mml:mover><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:munder><mml:mtext>E</mml:mtext><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003C9;</mml:mi></mml:mstyle><mml:mn>1</mml:mn></mml:msub><mml:mn>..</mml:mn><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003C9;</mml:mi></mml:mstyle><mml:mi>D</mml:mi></mml:msub><mml:mo>~</mml:mo><mml:msub><mml:mi>p</mml:mi><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle></mml:msub></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>b</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mn>..</mml:mn><mml:msub><mml:mi>b</mml:mi><mml:mi>D</mml:mi></mml:msub><mml:mo>~</mml:mo><mml:mi mathvariant='script'>U</mml:mi></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:munder><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mo>&#x02016;</mml:mo><mml:mi>&#x003A6;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>X</mml:mi></mml:mstyle><mml:mo>;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003A9;</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>b</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mover accent='true'><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='true'>&#x0005E;</mml:mo></mml:mover><mml:mo>&#x02212;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:msup><mml:mo>&#x02016;</mml:mo><mml:mn>2</mml:mn></mml:msup><mml:mo>+</mml:mo><mml:mo>&#x02016;</mml:mo><mml:mover accent='true'><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='true'>&#x0005E;</mml:mo></mml:mover><mml:msup><mml:mo>&#x02016;</mml:mo><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>=</mml:mo><mml:munder><mml:mtext>E</mml:mtext><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>&#x003F5;</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mn>..</mml:mn><mml:msub><mml:mi>&#x003F5;</mml:mi><mml:mi>D</mml:mi></mml:msub><mml:mo>~</mml:mo><mml:mi mathvariant='script'>N</mml:mi></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>b</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mn>..</mml:mn><mml:msub><mml:mi>b</mml:mi><mml:mi>D</mml:mi></mml:msub><mml:mo>~</mml:mo><mml:mi mathvariant='script'>U</mml:mi></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:munder><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mo>&#x02016;</mml:mo><mml:mi>&#x003A6;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>X</mml:mi></mml:mstyle><mml:mo>;</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>E</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>b</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mover accent='true'><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='true'>&#x0005E;</mml:mo></mml:mover><mml:mo>&#x02212;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:msup><mml:mo>&#x02016;</mml:mo><mml:mn>2</mml:mn></mml:msup><mml:mo>+</mml:mo><mml:mo>&#x02016;</mml:mo><mml:mover accent='true'><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='true'>&#x0005E;</mml:mo></mml:mover><mml:msup><mml:mo>&#x02016;</mml:mo><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <bold>E</bold> is the matrix containing the <italic>D</italic> random vectors <bold>&#x003F5;</bold><sub>1</sub>, &#x02026;, <bold>&#x003F5;</bold><sub><italic>D</italic></sub>. Thus, we can obtain an unbiased estimator of the encoding loss using a simple Monte Carlo estimator with a single sample. That is, if we sample <bold>E</bold> and <bold>b</bold> from their respective distributions, we obtain an unbiased estimator of <inline-formula><mml:math id="M27"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003B8;</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>:</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M28"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003B8;</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mo>||</mml:mo><mml:mi>&#x003A6;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003B8;</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>E</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:mo>-</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle><mml:mo>||</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mo>||</mml:mo><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:mo>||</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Note that it is possible to sample multiple noise matrices <bold>E</bold><sub>1</sub>, <bold>E</bold><sub>2</sub>, &#x02026; to lower the variance of the estimator, but in order to accelerate the computation we don&#x00027;t explore this alternative.</p></sec>
<sec>
<title>4.3 Adapt the encoder via generator training</title>
<p>We explore the parameter space of the generator <italic>f</italic><sub><bold>&#x003B8;</bold></sub> using stochastic gradient descent, where <inline-formula><mml:math id="M29"><mml:mstyle mathvariant="bold"><mml:mtext>&#x003B8;</mml:mtext></mml:mstyle><mml:mo>&#x02190;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>&#x003B8;</mml:mtext></mml:mstyle><mml:mo>-</mml:mo><mml:mi>&#x003B7;</mml:mi><mml:msub><mml:mrow><mml:mo>&#x02207;</mml:mo></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003B8;</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003B8;</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>. Its easy to show that <inline-formula><mml:math id="M30"><mml:msub><mml:mrow><mml:mo>&#x02207;</mml:mo></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003B8;</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mover accent="true"><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003B8;</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> is an unbiased estimator of the true gradient <inline-formula><mml:math id="M31"><mml:msub><mml:mrow><mml:mo>&#x02207;</mml:mo></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003B8;</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003B8;</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> because the expectation in <inline-formula><mml:math id="M32"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> does not depend on the parameters <bold>&#x003B8;</bold>. It is important to note that <inline-formula><mml:math id="M33"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is differentiable with respect to its parameters <bold>&#x003B8;</bold>. Indeed, every operation shown in <xref ref-type="disp-formula" rid="E8">Equation (8)</xref> is well behaved: <bold>&#x003A9;</bold> &#x0003D; <italic>f</italic><sub><bold>&#x003B8;</bold></sub>(<bold>E</bold>) is clearly differentiable, and so is <bold>Z</bold> &#x0003D; &#x003A6;(<bold>X</bold>; <bold>&#x003A9;</bold>, <bold>b</bold>) because <inline-formula><mml:math id="M34"><mml:mi>&#x003C6;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>&#x003A9;</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msqrt><mml:mrow><mml:mfrac><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:msqrt><mml:mo class="qopname">cos.</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003A9;</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>&#x0002B;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>.</p>
<p>Until here, we have covered how to parameterize and learn the sampling distribution. This equips FLASH with an encoding module well-adapted. Still, people may wonder if learning the encoding matrix <bold>&#x003A9;</bold> and offset <bold>b</bold> through gradient descent is a good alternative. We acknowledge that this might be a more direct measure, however, it will jeopardize the holographic property of HDC since it cannot guarantee that the encoding matrix is i.i.d. This means that the information in the encoded hypervector is no longer evenly distributed, and errors or noise in the encoding process will lead to higher performance loss due to the lack of hyperdimensional redundancy. Our proposed measure will ensure that FLASH will maintain the holographic HDC representation after tuning the encoder.</p></sec>
<sec>
<title>4.4 Balance the cost in training</title>
<p>The training in FLASH is a two-stage process where we first learn the generator <italic>f</italic><sub><bold>&#x003B8;</bold></sub>(<bold>&#x003F5;</bold>), i.e., in place of sampling from <italic>p</italic><sub><bold>&#x003B8;</bold></sub>(<bold>&#x003C9;</bold>)). Based on the first training stage, we then perform the model training that gives the regression hypervector <bold>w</bold>. In the second stage, the encoder will be generated using <italic>f</italic><sub><bold>&#x003B8;</bold></sub> and remain static as it has been optimized. Notice that the dimensionality <italic>D</italic> can be the same or different in these two training stages. As mentioned in the previous section, our optimization method for generator training is well-defined; but in practice, it can be further approximated to obtain faster convergence. Several algorithms have been widely used to accelerate the computation of ridge estimator (Paige and Saunders, <xref ref-type="bibr" rid="B26">1982</xref>; Defazio et al., <xref ref-type="bibr" rid="B5">2014</xref>). However, computing <inline-formula><mml:math id="M35"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003B8;</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> at every iteration for encoder learning ends up adding up the overhead. Recall that in prior HDC algorithms (Hernandez-Cane et al., <xref ref-type="bibr" rid="B10">2021</xref>; Hern&#x000E1;ndez-Cano et al., <xref ref-type="bibr" rid="B11">2021</xref>; Ni et al., <xref ref-type="bibr" rid="B23">2022a</xref>), <italic>D</italic> is supposed to be a high dimensionality such that model hypervectors have a larger capacity. In FLASH, we instead propose to decouple the high dimensionality requirement from the encoder/generator training since the generator <italic>f</italic><sub><bold>&#x003B8;</bold></sub> itself operates in &#x0211D;<sup><italic>M</italic></sup>. When training <italic>f</italic><sub><bold>&#x003B8;</bold></sub>, we encode data to &#x0211D;<sup><italic>D</italic></sup>&#x02032; instead, where <italic>D</italic>&#x02032; &#x0003C; <italic>D</italic> in order to accelerate the process. As for the regression process, we use a slightly larger dimensionality <italic>D</italic> for better regression accuracy. In fact, adapting the HDC encoder at first will also lower the requirement for model hypervector dimensionality <italic>D</italic> and thus reduce the training cost. Our results in the experiment show that FLASH, with a lower dimensionality, has comparable regression quality to the prior HDC method.</p></sec>
<sec>
<title>4.5 Time complexity</title>
<p>In this section, we discuss the time complexity of training our proposed method. Below we describe at a high level the steps required in our approach.</p>
<list list-type="simple">
<list-item><p>1. Train the surrogate sampling function <italic>f</italic><sub><bold>&#x003B8;</bold></sub>.</p></list-item></list>
<list list-type="simple">
<list-item><p>(a) Encode data to <italic>D</italic>&#x02032;-dimensional space.</p></list-item>
<list-item><p>(b) Compute the loss <inline-formula><mml:math id="M36"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003B8;</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>.</p></list-item>
<list-item><p>(c) Compute the gradient and update the parameters in <italic>f</italic><sub><bold>&#x003B8;</bold></sub>.</p></list-item>
<list-item><p>(d) Iterate this process until convergence.</p></list-item>
</list>
<list list-type="simple">
<list-item><p>2. Generate encoding matrix <bold>&#x003A9;</bold> using <italic>f</italic><sub><bold>&#x003B8;</bold></sub>.</p></list-item>
<list-item><p>3. Encode data to <italic>D</italic>-dimensional with the adapted HDC encoder.</p></list-item>
<list-item><p>4. Learn the regressing hypervector <bold>w</bold>.</p></list-item>
</list>
<p>In the first step, the overhead of computing <inline-formula><mml:math id="M37"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is considered minor as we encode data to <italic>D</italic>&#x02032;-dimensional space, limiting the cost of computing the estimator <inline-formula><mml:math id="M38"><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:math></inline-formula>. Generating random bases <bold>&#x003C9;</bold><sub><italic>i</italic></sub> &#x0003D; <italic>f</italic><sub><bold>&#x003B8;</bold></sub>(<bold>&#x003B5;</bold><sub><italic>i</italic></sub>) requires <italic>D</italic> forward passes of <italic>f</italic><sub><bold>&#x003B8;</bold></sub>, which will give a hyperdimensional mapping.</p>
<p>Encoding the data <inline-formula><mml:math id="M39"><mml:mstyle mathvariant="bold"><mml:mtext>Z</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mi>&#x003A6;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>&#x003A9;</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msqrt><mml:mrow><mml:mfrac><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:msqrt><mml:mo class="qopname">cos.</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003A9;</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, requires <inline-formula><mml:math id="M40"><mml:mrow><mml:mi mathvariant="script">O</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>N</mml:mi><mml:mi>M</mml:mi><mml:mi>D</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> operations, corresponding to the asymptotic of the most taxing operation - matrix multiplication of the <italic>N</italic>&#x000D7;<italic>M</italic> matrix <bold>X</bold> with a <italic>M</italic>&#x000D7;<italic>D</italic> matrix <bold>&#x003A9;</bold><sup><italic>T</italic></sup>, where <italic>N</italic> is the number of samples and <italic>M</italic> the number of features in original space.</p>
<p>The last step, computing <bold>w</bold> &#x0003D; (<bold>Z</bold><sup><italic>T</italic></sup><bold>Z</bold>&#x0002B;&#x003BB;<bold>I</bold>)&#x02212;1<bold>Z</bold><sup><italic>T</italic></sup><bold>y</bold> is, according to experimental evaluation, the most time-consuming step in our design. The theoretical time complexity is dominated by the <bold>Z</bold><sup><italic>T</italic></sup><bold>Z</bold> multiplication and the inverse operation, requiring <inline-formula><mml:math id="M41"><mml:mrow><mml:mi mathvariant="script">O</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>N</mml:mi><mml:msup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M42"><mml:mrow><mml:mi mathvariant="script">O</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> operations, respectively. Thankfully, the dampened linear least squares loss, <xref ref-type="disp-formula" rid="E3">Equation 3</xref>), has been heavily studied and several algorithms that approximate its solution exist such as LSQR (Paige and Saunders, <xref ref-type="bibr" rid="B26">1982</xref>). Moreover, the experimental evaluation suggests that with relatively small values of <italic>D</italic> (500 &#x02264; <italic>D</italic> &#x02264; 2000) we can obtain very accurate predictions (as shown in <xref ref-type="fig" rid="F5">Figure 5</xref>); with this configuration, we observe linear training time in the number of samples.</p>


</sec>


<sec>
<title>4.6 Formal derivation of encoding loss</title>
<p>In this section, we show that minimizing <inline-formula><mml:math id="M43"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> with respect to <bold>&#x003B8;</bold> is equivalent to maximizing a lower bound of the log-likelihood of the posterior distribution with joint parameters (<bold>&#x003B8;</bold>, <bold>w</bold>).</p>
<p>Recall that given a dataset of independent observations <inline-formula><mml:math id="M44"><mml:mrow><mml:mi mathvariant="script">D</mml:mi></mml:mrow></mml:math></inline-formula>, random parameters &#x003B8;&#x02208;&#x00398; with prior distribution &#x003B8;&#x0007E;<italic>p</italic>(&#x003B8;) the posterior distribution is <inline-formula><mml:math id="M45"><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003B8;</mml:mi><mml:mo>&#x02223;</mml:mo><mml:mrow><mml:mi mathvariant="script">D</mml:mi></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, which by Bayes&#x00027; Theorem can be expressed as <inline-formula><mml:math id="M46"><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003B8;</mml:mi><mml:mo>&#x02223;</mml:mo><mml:mrow><mml:mi mathvariant="script">D</mml:mi></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0221D;</mml:mo><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mi mathvariant="script">D</mml:mi></mml:mrow><mml:mo>&#x02223;</mml:mo><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, where <inline-formula><mml:math id="M47"><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mi mathvariant="script">D</mml:mi></mml:mrow><mml:mo>&#x02223;</mml:mo><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> is called the likelihood.</p>
<p>In regression analysis, we assume the relation <bold>y</bold> &#x0003D; <bold>Xw</bold>&#x0002B;<bold>&#x003B5;</bold>, where <bold>&#x003B5;</bold> is a random unobserved noise. Ordinary least squares regression sets the likelihood to be normally distributed <inline-formula><mml:math id="M48"><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle><mml:mo>&#x02223;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle><mml:mo>&#x0007E;</mml:mo><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mstyle mathvariant="bold"><mml:mtext>I</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>. As shown in <xref ref-type="disp-formula" rid="E9">Equation 9</xref>, maximizing the log-likelihood is equivalent to minimizing the squared error loss:</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M49"><mml:mrow><mml:mtable columnalign='left'><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:munder><mml:mrow><mml:mi>max</mml:mi></mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle></mml:munder><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo>&#x02223;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:munder><mml:mrow><mml:mi>max</mml:mi></mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle></mml:munder><mml:mi>ln</mml:mi><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo>&#x02223;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:munder><mml:mrow><mml:mi>max</mml:mi></mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle></mml:munder><mml:mi>ln</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>Z</mml:mi></mml:mfrac><mml:mi>exp</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mn>2</mml:mn></mml:mfrac><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo>&#x02212;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>X</mml:mi><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mi>T</mml:mi></mml:msup><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msup><mml:mi>&#x003C3;</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>I</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo>&#x02212;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>X</mml:mi><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>=</mml:mo><mml:munder><mml:mrow><mml:mi>max</mml:mi></mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle></mml:munder><mml:mo>&#x0007B;</mml:mo><mml:mi>ln</mml:mi><mml:mi>exp</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mn>2</mml:mn></mml:mfrac><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo>&#x02212;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>X</mml:mi><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mi>T</mml:mi></mml:msup><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msup><mml:mi>&#x003C3;</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>I</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo>&#x02212;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>X</mml:mi><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>ln</mml:mi><mml:mi>Z</mml:mi><mml:mo>&#x0007D;</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x02009;&#x02009;</mml:mtext><mml:mo>=</mml:mo><mml:munder><mml:mrow><mml:mi>min</mml:mi></mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle></mml:munder><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mn>2</mml:mn><mml:msup><mml:mi>&#x003C3;</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mfrac><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo>&#x02212;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>X</mml:mi><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mi>T</mml:mi></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo>&#x02212;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>X</mml:mi><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:munder><mml:mrow><mml:mi>min</mml:mi></mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle></mml:munder><mml:mo>&#x02016;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo>&#x02212;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>X</mml:mi><mml:mi>w</mml:mi></mml:mstyle><mml:msup><mml:mo>&#x02016;</mml:mo><mml:mn>2</mml:mn></mml:msup><mml:mo>.</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></disp-formula>
<p>In order to work in the high-dimensional space, we must add the encoding to the equation, or in this case the generator parameters <bold>&#x003B8;</bold>. We instead assume the relation <bold>y</bold> &#x0003D; <bold>Zw</bold>&#x0002B;<bold>&#x003B5;</bold>, where <bold>Z</bold> &#x0003D; &#x003A6;(<bold>X</bold>; <bold>&#x003A9;</bold>, <bold>b</bold>). Because the regression coefficients depend on <bold>&#x003A9;</bold> and <bold>b</bold>, we use the conditional likelihood <inline-formula><mml:math id="M50"><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle><mml:mo>&#x02223;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>&#x003A9;</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle><mml:mo>&#x0007E;</mml:mo><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>Z</mml:mtext></mml:mstyle><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mstyle mathvariant="bold"><mml:mtext>I</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>. We add the independence assumption between <bold>&#x003B8;</bold> and <bold>b</bold>. In <xref ref-type="disp-formula" rid="E10">Equation 10</xref>, the maximum log posterior is shown to be bounded using Jensen&#x00027;s inequality:</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M51"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:munder><mml:mrow><mml:mi>max</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle></mml:mrow></mml:munder><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>ln</mml:mi><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle><mml:mo>&#x02223;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:munder><mml:mrow><mml:mi>max</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle></mml:mrow></mml:munder><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>ln</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo>&#x02223;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>=</mml:mo><mml:munder><mml:mrow><mml:mi>max</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle></mml:mrow></mml:munder><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>ln</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:munder><mml:mtext>E</mml:mtext><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003C9;</mml:mi></mml:mstyle><mml:mn>1</mml:mn></mml:msub><mml:mn>..</mml:mn><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003C9;</mml:mi></mml:mstyle><mml:mi>D</mml:mi></mml:msub><mml:mo>~</mml:mo><mml:msub><mml:mi>p</mml:mi><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle></mml:msub></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>b</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mn>..</mml:mn><mml:msub><mml:mi>b</mml:mi><mml:mi>D</mml:mi></mml:msub><mml:mo>~</mml:mo><mml:mi mathvariant='script'>U</mml:mi></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:munder><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo>&#x02223;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003A9;</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>b</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>&#x02265;</mml:mo><mml:munder><mml:mrow><mml:mi>max</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle></mml:mrow></mml:munder><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:munder><mml:mtext>E</mml:mtext><mml:mrow><mml:msub><mml:mrow><mml:mtext>&#x000A0;</mml:mtext></mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003C9;</mml:mi></mml:mstyle><mml:mn>1</mml:mn></mml:msub><mml:mn>..</mml:mn><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003C9;</mml:mi></mml:mstyle><mml:mi>D</mml:mi></mml:msub><mml:mo>~</mml:mo><mml:msub><mml:mi>p</mml:mi><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle></mml:msub></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>b</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mn>..</mml:mn><mml:msub><mml:mi>b</mml:mi><mml:mi>D</mml:mi></mml:msub><mml:mo>~</mml:mo><mml:mi mathvariant='script'>U</mml:mi></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:msub></mml:mrow></mml:munder><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:munder><mml:munder><mml:mrow><mml:mi>ln</mml:mi><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo>&#x02223;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003A9;</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>b</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mo>+</mml:mo><mml:mi>ln</mml:mi><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo stretchy='true'>&#x0FE38;</mml:mo></mml:munder><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:msub><mml:mi>&#x02112;</mml:mi><mml:mi>R</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:munder><mml:mo>+</mml:mo><mml:mi>ln</mml:mi><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>=</mml:mo><mml:munder><mml:mrow><mml:mi>min</mml:mi></mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle></mml:munder><mml:munder><mml:mrow><mml:mi>min</mml:mi></mml:mrow><mml:mi>w</mml:mi></mml:munder><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:munder><mml:mtext>E</mml:mtext><mml:mrow><mml:msub><mml:mrow><mml:mtext>&#x000A0;</mml:mtext></mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003C9;</mml:mi></mml:mstyle><mml:mn>1</mml:mn></mml:msub><mml:mn>..</mml:mn><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003C9;</mml:mi></mml:mstyle><mml:mi>D</mml:mi></mml:msub><mml:mo>~</mml:mo><mml:msub><mml:mi>p</mml:mi><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle></mml:msub></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>b</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mn>..</mml:mn><mml:msub><mml:mi>b</mml:mi><mml:mi>D</mml:mi></mml:msub><mml:mo>~</mml:mo><mml:mi mathvariant='script'>U</mml:mi></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:msub></mml:mrow></mml:munder><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mi>&#x02112;</mml:mi><mml:mi>R</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>ln</mml:mi><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>=</mml:mo><mml:munder><mml:mrow><mml:mi>min</mml:mi></mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle></mml:munder><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:munder><mml:mtext>E</mml:mtext><mml:mrow><mml:msub><mml:mrow><mml:mtext>&#x000A0;</mml:mtext></mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003C9;</mml:mi></mml:mstyle><mml:mn>1</mml:mn></mml:msub><mml:mn>..</mml:mn><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003C9;</mml:mi></mml:mstyle><mml:mi>D</mml:mi></mml:msub><mml:mo>~</mml:mo><mml:msub><mml:mi>p</mml:mi><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle></mml:msub></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>b</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mn>..</mml:mn><mml:msub><mml:mi>b</mml:mi><mml:mi>D</mml:mi></mml:msub><mml:mo>~</mml:mo><mml:mi mathvariant='script'>U</mml:mi></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:msub></mml:mrow></mml:munder><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mi>&#x02112;</mml:mi><mml:mi>R</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mover accent='true'><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>w</mml:mi></mml:mstyle><mml:mo stretchy='true'>&#x0005E;</mml:mo></mml:mover><mml:mo stretchy='false'>)</mml:mo><mml:mo>+</mml:mo><mml:mi>R</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:munder><mml:mrow><mml:mi>min</mml:mi></mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle></mml:munder><mml:msub><mml:mi>&#x02112;</mml:mi><mml:mi>E</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec></sec>
<sec id="s5">
<title>5 Experimental results</title>
<sec>
<title>5.1 Experimental settings</title>
<p>We implement the proposed design using Python on the Intel Core i7-12700K CPU platform. The core process of adapting the encoder is implemented using Pytorch, and the regression process is based on the implementation provided by Scikit-Learn. We evaluate the accuracy of our design on several practical regression datasets listed in <xref ref-type="table" rid="T1">Table 1</xref>, with up to 20,000 samples and 80 features. <xref ref-type="table" rid="T2">Table 2</xref> describes the baseline models used to compare with our design, including ridge regression that also leverages RFF approximation and the previous state-of-the-art HDC-based regression algorithm RegHD. During the experiments, we test two settings for our design (FLASH and A-FLASH), slightly different in model size and dimensionality. The name of the second setting stands for &#x0201C;Accurate FLASH,&#x0201D; which has larger dimensionality and model size. In <xref ref-type="table" rid="T3">Table 3</xref>, we provide the hyperparameters used for the different regression models in our experiments.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Various regression datasets available on OpenML (Vanschoren et al., <xref ref-type="bibr" rid="B38">2013</xref>).</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="center"><bold><italic>N</italic></bold></th>
<th valign="top" align="center"><bold><italic>M</italic></bold></th>
<th valign="top" align="left"><bold>Description</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">kin8nm</td>
<td valign="top" align="center">8,192</td>
<td valign="top" align="center">9</td>
<td valign="top" align="left">Forward kinematics of an 8 link robot arm</td>
</tr> <tr>
<td valign="top" align="left">MiamiHousing2016</td>
<td valign="top" align="center">13,932</td>
<td valign="top" align="center">17</td>
<td valign="top" align="left">Sale price of houses</td>
</tr> <tr>
<td valign="top" align="left">pol</td>
<td valign="top" align="center">15,000</td>
<td valign="top" align="center">49</td>
<td valign="top" align="left">Telecommunication problem</td>
</tr> <tr>
<td valign="top" align="left">Houses</td>
<td valign="top" align="center">20,640</td>
<td valign="top" align="center">9</td>
<td valign="top" align="left">Predict house value</td>
</tr>
<tr>
<td valign="top" align="left">Superconduct</td>
<td valign="top" align="center">21,263</td>
<td valign="top" align="center">82</td>
<td valign="top" align="left">Predict critical temperature of superconductors</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>N, Number of samples; M, number of features.</p>
</table-wrap-foot>
</table-wrap>

<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Regression models used in experiments: the last two are our proposed design, and the rest are baseline methods.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="left"><bold>Description</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">SVR</td>
<td valign="top" align="left">Support vector regression with RBF kernel</td>
</tr> <tr>
<td valign="top" align="left">Kernel Ridge</td>
<td valign="top" align="left">Analytical solution of kernel regression with RBF kernel</td>
</tr> <tr>
<td valign="top" align="left">Linear Regression</td>
<td valign="top" align="left">Ordinary least squares</td>
</tr> <tr>
<td valign="top" align="left">RFF &#x0002B; Ridge</td>
<td valign="top" align="left">RFF approximation of RBF kernel, followed by ridge regression</td>
</tr> <tr>
<td valign="top" align="left">RegHD</td>
<td valign="top" align="left">HDC regression based on VFA (Hern&#x000E1;ndez-Cano et al., <xref ref-type="bibr" rid="B11">2021</xref>)</td>
</tr> <tr>
<td valign="top" align="left">FLASH</td>
<td valign="top" align="left">Our design optimized for better inference runtime</td>
</tr>
<tr>
<td valign="top" align="left">A-FLASH</td>
<td valign="top" align="left">Our design optimized for better regression accuracy</td>
</tr></tbody>
</table>
</table-wrap>

<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Hyperparameters used for the regression models.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="left"><bold>Hyperparameter</bold></th>
<th valign="top" align="left"><bold>Value</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">FLASH <break/> (A-FLASH)</td>
<td valign="top" align="left"><italic>D</italic></td>
<td valign="top" align="left">500 (resp. 2000)</td>
</tr>
<tr>
<td/>
<td valign="top" align="left"><italic>D</italic>&#x02032;</td>
<td valign="top" align="left">75 (resp. 250)</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x003B1;<sup>&#x02020;</sup></td>
<td valign="top" align="left">0.01 &#x02264; &#x003B1; &#x02264; 0.1</td>
</tr>
<tr>
<td/>
<td valign="top" align="left"><italic>f</italic><sub><bold>&#x003B8;</bold></sub> layers</td>
<td valign="top" align="left">[32, 32] (resp. [64, 64, 64, 64])</td>
</tr>
<tr>
<td/>
<td valign="top" align="left"><italic>f</italic><sub><bold>&#x003B8;</bold></sub> activation</td>
<td valign="top" align="left">tanh</td>
</tr>
<tr>
<td/>
<td valign="top" align="left"><italic>f</italic><sub><bold>&#x003B8;</bold></sub> learning rate<sup>&#x02020;</sup></td>
<td valign="top" align="left">0.001 &#x02264; lr &#x02264; 0.01</td>
</tr> <tr>
<td valign="top" align="left">&#x0002A;RegHD</td>
<td valign="top" align="left"><italic>D</italic></td>
<td valign="top" align="left">2000</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">Number of models</td>
<td valign="top" align="left">1</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">Learning rate</td>
<td valign="top" align="left">0.035</td>
</tr> <tr>
<td valign="top" align="left">RFF &#x0002B; Ridge</td>
<td valign="top" align="left"><italic>D</italic></td>
<td valign="top" align="left">2000</td>
</tr> <tr>
<td valign="top" align="left">&#x0002A;SVR</td>
<td valign="top" align="left"><italic>C</italic></td>
<td valign="top" align="left">0.1 &#x02264; <italic>C</italic> &#x02264; 100</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x003B5;</td>
<td valign="top" align="left">0.01 &#x02264; &#x003B5; &#x02264; 1</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x003B3;</td>
<td valign="top" align="left"><inline-formula><mml:math id="M52"><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mtext>n</mml:mtext><mml:mstyle class="text"><mml:mtext>_</mml:mtext></mml:mstyle><mml:mtext>feature</mml:mtext><mml:msup><mml:mrow><mml:mtext>s</mml:mtext></mml:mrow><mml:mrow><mml:mo>&#x0002A;</mml:mo></mml:mrow></mml:msup><mml:mtext>Var</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:math></inline-formula></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p><sup>&#x02020;</sup>This hyperparameter was tuned using grid search. <sup>&#x0002A;</sup>SVR means support vector regression and RegHD is the name of a prior HDC-based regression framework.</p>
</table-wrap-foot>
</table-wrap></sec>




<sec>
<title>5.2 Performance on synthetic data</title>
<p>In this section, we analyze the performance of the proposed design in custom 1D regression problems of the form <italic>y</italic> &#x0003D; <italic>f</italic>(<italic>x</italic>)&#x0002B;&#x003B5; with <inline-formula><mml:math id="M53"><mml:mi>&#x003B5;</mml:mi><mml:mo>&#x0007E;</mml:mo><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> and different choices of target function <italic>f</italic>. In <xref ref-type="fig" rid="F2">Figure 2</xref>, we also show the encoder&#x00027;s probability distribution <italic>p</italic><sub><bold>&#x003B8;</bold></sub>(<bold>&#x003C9;</bold>) learnt in the process and the actual kernel function <italic>K</italic>(&#x00394;) &#x0003D; E[&#x003D5;(<bold>x</bold>)<sup><italic>T</italic></sup>&#x003D5;(<bold>0</bold>)]. For comparison, RBF kernel has associated a Gaussian <inline-formula><mml:math id="M54"><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>0</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow></mml:mfrac><mml:mstyle mathvariant="bold"><mml:mtext>I</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> distribution.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>The <bold>top row</bold> corresponds to the data created, in green is the training data, in orange is the prediction of FLASH, and in blue is the prediction of SVR. The <bold>second row</bold> corresponds to the distribution <italic>p</italic><sub><bold>&#x003B8;</bold></sub>(<bold>&#x003C9;</bold>) of the random bases in the encoder. The <bold>last row</bold> shows the (approximate) kernel associated with the distribution as in <xref ref-type="disp-formula" rid="E1">Equation (1)</xref>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1371988-g0002.tif"/>
</fig>


<p>From this experiment, we conclude that our optimization proposal for the encoder loss <inline-formula><mml:math id="M55"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> works well in practice and the shapes of learned distributions are varied for each dataset. Moreover, we observe that our proposal adapts to different scales in the data making a clear distinction with SVR. For instance, the first predicted function <italic>f</italic>(<italic>x</italic>) &#x0003D; 10<italic>x</italic><sup>2</sup>, where SVR clearly underperforms where |<italic>x</italic>|&#x0003E;1.1.</p></sec>
<sec>
<title>5.3 Regression quality and efficiency comparison</title>
<p>In this section, we compare the performance of FLASH (as well as A-FLASH) against several baseline regression algorithms using multiple regression datasets. We perform 5 times repeated 5-fold cross-validation in each dataset and report the average prediction quality, confidence intervals, statistical tests for significance, and runtime taken for each fold. We select the most important hyperparameters in SVR and our design using grid search. Our results are summarized in <xref ref-type="fig" rid="F3">Figure 3</xref>.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Comparison of our approach against other methods in several datasets. The <bold>top plot</bold> shows the prediction quality using the <italic>r</italic><sup>2</sup> metric (larger is better). The <bold>bottom plot</bold> shows the inference time. We exclude the linear regression runtime for better visualization since the value is relatively small. Note that SVR performs poorly on the MiamiHousing dataset even after grid search (thus not visible in the figure) and the log scale is used in the bottom plot. We report the &#x000B1;95% confidence interval and use the Nadeau and Bengio&#x00027;s corrected <italic>t</italic>-test (Nadeau and Bengio, <xref ref-type="bibr" rid="B20">1999</xref>) for significance in prediction quality comparison: <sup>&#x0002A;&#x0002A;&#x0002A;</sup><italic>P</italic> &#x0003C; 0.001, <sup>&#x0002A;&#x0002A;</sup><italic>P</italic> &#x0003C; 0.01, <sup>&#x0002A;</sup><italic>P</italic> &#x0003C; 0.05; NS, not significant.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1371988-g0003.tif"/>
</fig>


<p>We observe that our approach is always comparable in accuracy with other state-of-the-art approaches. The accurate version of our approach (A-FLASH) is consistently ranked at the top. Particularly, because our encoder is learnable and well-adapted, we are generally more accurate than other algorithms leveraging static encoder or fixed kernel. In comparison with the prior HDC-based method, A-FLASH achieves significantly better quality without adding notable overhead for inference. In addition, the fast version of our approach (FLASH) is generally among the fastest models. During inference, it is faster than other baselines, including classical kernel-based approaches such as SVR and Kernel Ridge. This is because our prediction complexity is constant with respect to the number of samples. On average, FLASH is about 3.7 &#x000D7; faster inference than the RegHD, 5.5 &#x000D7; faster than kernel ridge/RFF ridge, and 13.75 &#x000D7; faster than SVR.</p></sec>
<sec>
<title>5.4 Scalability results</title>
<p>In this section, we create the Friedman regression datasets (Friedman, <xref ref-type="bibr" rid="B8">1991</xref>) with an increasing number of samples to test the scalability of the proposed algorithm and compare it with other approaches. We observe that our approach is well-suited for large-scale data as we have a linear trend in the time taken to train and also inference time. Meanwhile, the time taken to train classical kernel approaches such as SVR and kernel ridge grows noticeably faster due to their higher computational complexity. Our results are summarized in <xref ref-type="fig" rid="F4">Figure 4</xref>. The leftmost plot shows that our approach is the fastest to achieve high prediction quality even with a small number of samples; in fact, FLASH constantly achieves better accuracy when the training set grows. In terms of inference speed, FLASH is about twice as fast as the other approaches with 5000 training samples, and the gap in between continues to expand.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>The <bold>left graph plots</bold> the prediction quality (root MSE) against the number of samples in different regression models. The <bold>second and third plot</bold> shows the time taken to train and do inference on a logarithmic scale, respectively. Note the slower growth rate of our approach compared with other kernel approaches.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1371988-g0004.tif"/>
</fig></sec>


<sec>
<title>5.5 Impact of dimensionality</title>
<p>In this section, we explore the impact of dimensionality (<italic>D</italic>) in our design for various datasets. <xref ref-type="fig" rid="F5">Figure 5</xref> displays our results in terms of prediction quality (MSE) and time taken to train the model for different values of <italic>D</italic>. In the section on &#x0201C;Time Complexity,&#x0201D; we derived the time complexity of our approach to be <inline-formula><mml:math id="M56"><mml:mrow><mml:mi mathvariant="script">O</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>N</mml:mi><mml:msup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, which is consistent with our experimental results. However, it is worth mentioning that even for relatively small dimensionality (e.g., <italic>D</italic> &#x0003D; 500) the gain of accuracy for further increasing dimensionality is not significant. Thus, even if the theoretical complexity of the approach is large, in practice, we can obtain acceptable results rapidly.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Impact of dimensionality. Each graph shows the trade-off obtained when increasing the dimensionality <italic>D</italic>. The metrics measured are the coefficient of determination (<italic>r</italic><sup>2</sup>), training time, and inference time, respectively across the three plots. Notice the fast convergence of accuracy even with low dimensionality and the approximately linear scale in the time measured.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1371988-g0005.tif"/>
</fig>
</sec></sec>
<sec sec-type="conclusions" id="s6">
<title>6 Conclusion</title>
<p>In this paper, we present a novel HDC algorithm that features an adaptive and learnable encoder design. Unlike previous HDC works that solely focus on the learning of model hypervectors, our work also aims at providing a hyperdimensional representation that is more suitable to current tasks. Instead of learning the encoder directly, we construct a parameterized distribution that helps preserve the holographic property of HDC encoding. The results of several regression tasks show that our proposed algorithm can significantly boost the accuracy, surpassing the existing HDC-based arts and providing lower inference time.</p></sec>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://www.openml.org/search?type=data">https://www.openml.org/search?type=data</ext-link>.</p></sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>AH-C: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Project administration, Software, Supervision, Validation, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. YN: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Software, Validation, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. ZZ: Conceptualization, Formal analysis, Investigation, Methodology, Validation, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. AZ: Data curation, Formal analysis, Investigation, Methodology, Validation, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. MI: Conceptualization, Data curation, Formal analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing.</p></sec>
</body>
<back>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was supported in part by DARPA Young Faculty Award, National Science Foundation &#x00023;2127780, &#x00023;2319198, &#x00023;2321840, &#x00023;2312517, and &#x00023;2235472, Semiconductor Research Corporation (SRC), Office of Naval Research through the Young Investigator Program Award, and grants &#x00023;N00014-21-1-2225 and &#x00023;N00014-22-1-2067, the Air Force Office of Scientific Research, grants &#x00023;FA9550-22-1-0253, and generous gifts from Cisco. This study received funding from SRC. The funders were not involved in the study design, collection, analysis, interpretation of data, the writing of this article or the decision to submit it for publication.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Barkam</surname> <given-names>H. E.</given-names></name> <name><surname>Jeon</surname> <given-names>S. E.</given-names></name> <name><surname>Yun</surname> <given-names>S.</given-names></name> <name><surname>Yeung</surname> <given-names>C.</given-names></name> <name><surname>Zou</surname> <given-names>Z.</given-names></name> <name><surname>Jiao</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2023a</year>). &#x0201C;Hyperdimensional computing for resilient edge learning,&#x0201D; <italic>in 2023 IEEE/ACM International Conference on Computer Aided Design (ICCAD)</italic> (IEEE), <fpage>1</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1109/ICCAD57390.2023.10323671</pub-id></citation>
</ref>
<ref id="B2">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Barkam</surname> <given-names>H. E.</given-names></name> <name><surname>Yun</surname> <given-names>S.</given-names></name> <name><surname>Genssler</surname> <given-names>P. R.</given-names></name> <name><surname>Zou</surname> <given-names>Z.</given-names></name> <name><surname>Liu</surname> <given-names>C.-K.</given-names></name> <name><surname>Amrouch</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2023b</year>). <article-title>&#x0201C;Hdgim: hyperdimensional genome sequence matching on unreliable highly scaled fefet,&#x0201D;</article-title> in <source>2023 Design, Automation &#x00026;Test in Europe Conference &#x00026;Exhibition (DATE)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>1</fpage>&#x02013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.23919/DATE56975.2023.10137331</pub-id></citation>
</ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Issa</surname> <given-names>M.</given-names></name> <name><surname>Ni</surname> <given-names>Y.</given-names></name> <name><surname>Imani</surname> <given-names>M.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Darl: distributed reconfigurable accelerator for hyperdimensional reinforcement learning,&#x0201D;</article-title> in <source>Proceedings of the 41st IEEE/ACM International Conference on Computer-Aided Design</source>, 1&#x02013;9. <pub-id pub-id-type="doi">10.1145/3508352.3549437</pub-id></citation>
</ref>
<ref id="B4">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Zakeri</surname> <given-names>A.</given-names></name> <name><surname>Wen</surname> <given-names>F.</given-names></name> <name><surname>Barkam</surname> <given-names>H. E.</given-names></name> <name><surname>Imani</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Hypergraf: Hyperdimensional graph-based reasoning acceleration on fpga,&#x0201D;</article-title> in <source>2023 33rd International Conference on Field-Programmable Logic and Applications (FPL)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>34</fpage>&#x02013;<lpage>41</lpage>. <pub-id pub-id-type="doi">10.1109/FPL60245.2023.00013</pub-id></citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Defazio</surname> <given-names>A.</given-names></name> <name><surname>Bach</surname> <given-names>F.</given-names></name> <name><surname>Lacoste-Julien</surname> <given-names>S.</given-names></name></person-group> (<year>2014</year>). <article-title>&#x0201C;Saga: a fast incremental gradient method with support for non-strongly convex composite objectives,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source> 27.</citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Frady</surname> <given-names>E. P.</given-names></name> <name><surname>Kleyko</surname> <given-names>D.</given-names></name> <name><surname>Kymn</surname> <given-names>C. J.</given-names></name> <name><surname>Olshausen</surname> <given-names>B. A.</given-names></name> <name><surname>Sommer</surname> <given-names>F. T.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Computing on functions using randomized vector representations (in brief),&#x0201D;</article-title> in <source>Proceedings of the 2022 Annual Neuro-Inspired Computational Elements Conference</source> 115&#x02013;122. <pub-id pub-id-type="doi">10.1145/3517343.3522597</pub-id></citation>
</ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Frady</surname> <given-names>E. P.</given-names></name> <name><surname>Kleyko</surname> <given-names>D.</given-names></name> <name><surname>Sommer</surname> <given-names>F. T.</given-names></name></person-group> (<year>2021</year>). <article-title>Variable binding for sparse distributed representations: theory and applications</article-title>. <source>IEEE Trans. Neural Netw. Learn. Syst</source>. <volume>34</volume>, <fpage>2191</fpage>&#x02013;<lpage>2204</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2021.3105949</pub-id><pub-id pub-id-type="pmid">34478381</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Friedman</surname> <given-names>J. H.</given-names></name></person-group> (<year>1991</year>). <article-title>Multivariate adaptive regression splines</article-title>. <source>Ann. Statist</source>. <volume>19</volume>, <fpage>1</fpage>&#x02013;<lpage>67</lpage>. <pub-id pub-id-type="doi">10.1214/aos/1176347963</pub-id></citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ge</surname> <given-names>L.</given-names></name> <name><surname>Parhi</surname> <given-names>K. K.</given-names></name></person-group> (<year>2020</year>). <article-title>Classification using hyperdimensional computing: a review</article-title>. <source>IEEE Circ. Syst. Magaz</source>. <volume>20</volume>, <fpage>30</fpage>&#x02013;<lpage>47</lpage>. <pub-id pub-id-type="doi">10.1109/MCAS.2020.2988388</pub-id></citation>
</ref>
<ref id="B10">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Hernandez-Cane</surname> <given-names>A.</given-names></name> <name><surname>Matsumoto</surname> <given-names>N.</given-names></name> <name><surname>Ping</surname> <given-names>E.</given-names></name> <name><surname>Imani</surname> <given-names>M.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Onlinehd: robust, efficient, and single-pass online learning using hyperdimensional system,&#x0201D;</article-title> in <source>2021 Design, Automation &#x00026;Test in Europe Conference &#x00026;Exhibition (DATE)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>56</fpage>&#x02013;<lpage>61</lpage>. <pub-id pub-id-type="doi">10.23919/DATE51398.2021.9474107</pub-id></citation>
</ref>
<ref id="B11">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Hern&#x000E1;ndez-Cano</surname> <given-names>A.</given-names></name> <name><surname>Zhuo</surname> <given-names>C.</given-names></name> <name><surname>Yin</surname> <given-names>X.</given-names></name> <name><surname>Imani</surname> <given-names>M.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Reghd: robust and efficient regression in hyper-dimensional learning system,&#x0201D;</article-title> in <source>2021 58th ACM/IEEE Design Automation Conference (DAC)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>7</fpage>&#x02013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1109/DAC18074.2021.9586284</pub-id></citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hersche</surname> <given-names>M.</given-names></name> <name><surname>Zeqiri</surname> <given-names>M.</given-names></name> <name><surname>Benini</surname> <given-names>L.</given-names></name> <name><surname>Sebastian</surname> <given-names>A.</given-names></name> <name><surname>Rahimi</surname> <given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>A neuro-vector-symbolic architecture for solving raven&#x00027;s progressive matrices</article-title>. <source>Nat. Mach. Intell</source>. <volume>5</volume>, <fpage>363</fpage>&#x02013;<lpage>375</lpage>. <pub-id pub-id-type="doi">10.1038/s42256-023-00630-8</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Imani</surname> <given-names>M.</given-names></name> <name><surname>Morris</surname> <given-names>J.</given-names></name> <name><surname>Messerly</surname> <given-names>J.</given-names></name> <name><surname>Shu</surname> <given-names>H.</given-names></name> <name><surname>Deng</surname> <given-names>Y.</given-names></name> <name><surname>Rosing</surname> <given-names>T.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Bric: locality-based encoding for energy-efficient brain-inspired hyperdimensional computing,&#x0201D;</article-title> in <source>Proceedings of the 56th Annual Design Automation Conference 2019</source>, 1&#x02013;6. <pub-id pub-id-type="doi">10.1145/3316781.3317785</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Issa</surname> <given-names>M.</given-names></name> <name><surname>Shahhosseini</surname> <given-names>S.</given-names></name> <name><surname>Ni</surname> <given-names>Y.</given-names></name> <name><surname>Hu</surname> <given-names>T.</given-names></name> <name><surname>Abraham</surname> <given-names>D.</given-names></name> <name><surname>Rahmani</surname> <given-names>A. M.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>&#x0201C;Hyperdimensional hybrid learning on end-edge-cloud networks,&#x0201D;</article-title> in <source>2022 IEEE 40th International Conference on Computer Design (ICCD)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>652</fpage>&#x02013;<lpage>655</lpage>. <pub-id pub-id-type="doi">10.1109/ICCD56317.2022.00100</pub-id></citation>
</ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kanerva</surname> <given-names>P.</given-names></name></person-group> (<year>2009</year>). <article-title>Hyperdimensional computing: an introduction to computing in distributed representation with high-dimensional random vectors</article-title>. <source>Cogn. Comput</source>. <volume>1</volume>, <fpage>139</fpage>&#x02013;<lpage>159</lpage>. <pub-id pub-id-type="doi">10.1007/s12559-009-9009-8</pub-id></citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Karunaratne</surname> <given-names>G.</given-names></name> <name><surname>Le Gallo</surname> <given-names>M.</given-names></name> <name><surname>Cherubini</surname> <given-names>G.</given-names></name> <name><surname>Benini</surname> <given-names>L.</given-names></name> <name><surname>Rahimi</surname> <given-names>A.</given-names></name> <name><surname>Sebastian</surname> <given-names>A.</given-names></name></person-group> (<year>2020</year>). <article-title>In-memory hyperdimensional computing</article-title>. <source>Nat. Electr</source>. <volume>3</volume>, <fpage>327</fpage>&#x02013;<lpage>337</lpage>. <pub-id pub-id-type="doi">10.1038/s41928-020-0410-3</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kleyko</surname> <given-names>D.</given-names></name> <name><surname>Davies</surname> <given-names>M.</given-names></name> <name><surname>Frady</surname> <given-names>E. P.</given-names></name> <name><surname>Kanerva</surname> <given-names>P.</given-names></name> <name><surname>Kent</surname> <given-names>S. J.</given-names></name> <name><surname>Olshausen</surname> <given-names>B. A.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Vector symbolic architectures as a computing framework for nanoscale hardware</article-title>. <source>arXiv preprint arXiv:2106.05268</source>.<pub-id pub-id-type="pmid">37868615</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kleyko</surname> <given-names>D.</given-names></name> <name><surname>Rachkovskij</surname> <given-names>D.</given-names></name> <name><surname>Osipov</surname> <given-names>E.</given-names></name> <name><surname>Rahimi</surname> <given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>A survey on hyperdimensional computing aka vector symbolic architectures, part ii: applications, cognitive models, and challenges</article-title>. <source>ACM Comput. Surv</source>. <volume>55</volume>, <fpage>1</fpage>&#x02013;<lpage>52</lpage>. <pub-id pub-id-type="doi">10.1145/3558000</pub-id></citation>
</ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kleyko</surname> <given-names>D.</given-names></name> <name><surname>Rahimi</surname> <given-names>A.</given-names></name> <name><surname>Rachkovskij</surname> <given-names>D. A.</given-names></name> <name><surname>Osipov</surname> <given-names>E.</given-names></name> <name><surname>Rabaey</surname> <given-names>J. M.</given-names></name></person-group> (<year>2018</year>). <article-title>Classification and recall with binary hyperdimensional computing: tradeoffs in choice of density and mapping characteristics</article-title>. <source>IEEE Trans. Neural Netw. Learn. Syst</source>. <volume>29</volume>, <fpage>5880</fpage>&#x02013;<lpage>5898</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2018.2814400</pub-id><pub-id pub-id-type="pmid">29993669</pub-id></citation></ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nadeau</surname> <given-names>C.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name></person-group> (<year>1999</year>). <article-title>&#x0201C;Inference for the generalization error,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source> 12.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ni</surname> <given-names>Y.</given-names></name> <name><surname>Abraham</surname> <given-names>D.</given-names></name> <name><surname>Issa</surname> <given-names>M.</given-names></name> <name><surname>Kim</surname> <given-names>Y.</given-names></name> <name><surname>Mercati</surname> <given-names>P.</given-names></name> <name><surname>Imani</surname> <given-names>M.</given-names></name></person-group> (<year>2023a</year>). <article-title>&#x0201C;Efficient off-policy reinforcement learning via brain-inspired computing,&#x0201D;</article-title> in <source>Proceedings of the Great Lakes Symposium on VLSI 2023</source>, 449&#x02013;453. <pub-id pub-id-type="doi">10.1145/3583781.3590298</pub-id></citation>
</ref>
<ref id="B22">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ni</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Poduval</surname> <given-names>P.</given-names></name> <name><surname>Zou</surname> <given-names>Z.</given-names></name> <name><surname>Mercati</surname> <given-names>P.</given-names></name> <name><surname>Imani</surname> <given-names>M.</given-names></name></person-group> (<year>2023b</year>). <article-title>&#x0201C;Brain-inspired trustworthy hyperdimensional computing with efficient uncertainty quantification,&#x0201D;</article-title> in <source>2023 IEEE/ACM International Conference on Computer Aided Design (ICCAD)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>01</fpage>&#x02013;<lpage>09</lpage>. <pub-id pub-id-type="doi">10.1109/ICCAD57390.2023.10323657</pub-id></citation>
</ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ni</surname> <given-names>Y.</given-names></name> <name><surname>Issa</surname> <given-names>M.</given-names></name> <name><surname>Abraham</surname> <given-names>D.</given-names></name> <name><surname>Imani</surname> <given-names>M.</given-names></name> <name><surname>Yin</surname> <given-names>X.</given-names></name> <name><surname>Imani</surname> <given-names>M.</given-names></name></person-group> (<year>2022a</year>). <article-title>&#x0201C;Hdpg: Hyperdimensional policy-based reinforcement learning for continuous control,&#x0201D;</article-title> in <source>Proceedings of the 59th ACM/IEEE Design Automation Conference</source> 1141&#x02013;1146. <pub-id pub-id-type="doi">10.1145/3489517.3530668</pub-id></citation>
</ref>
<ref id="B24">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ni</surname> <given-names>Y.</given-names></name> <name><surname>Kim</surname> <given-names>Y.</given-names></name> <name><surname>Rosing</surname> <given-names>T.</given-names></name> <name><surname>Imani</surname> <given-names>M.</given-names></name></person-group> (<year>2022b</year>). <article-title>&#x0201C;Algorithm-hardware co-design for efficient brain-inspired hyperdimensional learning on edge,&#x0201D;</article-title> in <source>2022 Design, Automation &#x00026;Test in Europe Conference &#x00026;Exhibition (DATE)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>292</fpage>&#x02013;<lpage>297</lpage>. <pub-id pub-id-type="doi">10.23919/DATE54114.2022.9774524</pub-id></citation>
</ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ni</surname> <given-names>Y.</given-names></name> <name><surname>Lesica</surname> <given-names>N.</given-names></name> <name><surname>Zeng</surname> <given-names>F.-G.</given-names></name> <name><surname>Imani</surname> <given-names>M.</given-names></name></person-group> (<year>2022c</year>). <article-title>&#x0201C;Neurally-inspired hyperdimensional classification for efficient and robust biosignal processing,&#x0201D;</article-title> in <source>Proceedings of the 41st IEEE/ACM International Conference on Computer-Aided Design</source> 1&#x02013;9. <pub-id pub-id-type="doi">10.1145/3508352.3549477</pub-id></citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Paige</surname> <given-names>C. C.</given-names></name> <name><surname>Saunders</surname> <given-names>M. A.</given-names></name></person-group> (<year>1982</year>). <article-title>Lsqr: an algorithm for sparse linear equations and sparse least squares</article-title>. <source>ACM Trans. Mathem. Softw</source>. <volume>8</volume>, <fpage>43</fpage>&#x02013;<lpage>71</lpage>. <pub-id pub-id-type="doi">10.1145/355984.355989</pub-id></citation>
</ref>
<ref id="B27">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Pale</surname> <given-names>U.</given-names></name> <name><surname>Teijeiro</surname> <given-names>T.</given-names></name> <name><surname>Atienza</surname> <given-names>D.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Exg signal feature selection using hyperdimensional computing encoding,&#x0201D;</article-title> in <source>2022 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>1688</fpage>&#x02013;<lpage>1693</lpage>. <pub-id pub-id-type="doi">10.1109/BIBM55620.2022.9995107</pub-id></citation>
</ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Park</surname> <given-names>J.</given-names></name> <name><surname>Sandberg</surname> <given-names>I. W.</given-names></name></person-group> (<year>1991</year>). <article-title>Universal approximation using radial-basis-function networks</article-title>. <source>Neur. Comput</source>. <volume>3</volume>, <fpage>246</fpage>&#x02013;<lpage>257</lpage>. <pub-id pub-id-type="doi">10.1162/neco.1991.3.2.246</pub-id><pub-id pub-id-type="pmid">31167308</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Poduval</surname> <given-names>P.</given-names></name> <name><surname>Alimohamadi</surname> <given-names>H.</given-names></name> <name><surname>Zakeri</surname> <given-names>A.</given-names></name> <name><surname>Imani</surname> <given-names>F.</given-names></name> <name><surname>Najafi</surname> <given-names>M. H.</given-names></name> <name><surname>Givargis</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2022a</year>). <article-title>Graphd: Graph-based hyperdimensional memorization for brain-like cognitive learning</article-title>. <source>Front. Neurosci</source>. <volume>16</volume>:<fpage>757125</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2022.757125</pub-id><pub-id pub-id-type="pmid">35185456</pub-id></citation></ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Poduval</surname> <given-names>P.</given-names></name> <name><surname>Ni</surname> <given-names>Y.</given-names></name> <name><surname>Kim</surname> <given-names>Y.</given-names></name> <name><surname>Ni</surname> <given-names>K.</given-names></name> <name><surname>Kumar</surname> <given-names>R.</given-names></name> <name><surname>Cammarota</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2022b</year>). <article-title>&#x0201C;Adaptive neural recovery for highly robust brain-like representation,&#x0201D;</article-title> in <source>Proceedings of the 59th ACM/IEEE Design Automation Conference</source> 367&#x02013;372. <pub-id pub-id-type="doi">10.1145/3489517.3530659</pub-id></citation>
</ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rachkovskij</surname> <given-names>D.</given-names></name></person-group> (<year>2015</year>). <article-title>Formation of similarity-reflecting binary vectors with random binary projections</article-title>. <source>Cybern. Syst. Analy</source>. <volume>51</volume>, <fpage>313</fpage>&#x02013;<lpage>323</lpage>. <pub-id pub-id-type="doi">10.1007/s10559-015-9723-z</pub-id></citation>
</ref>
<ref id="B32">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Rahimi</surname> <given-names>A.</given-names></name> <name><surname>Benatti</surname> <given-names>S.</given-names></name> <name><surname>Kanerva</surname> <given-names>P.</given-names></name> <name><surname>Benini</surname> <given-names>L.</given-names></name> <name><surname>Rabaey</surname> <given-names>J. M.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Hyperdimensional biosignal processing: a case study for emg-based hand gesture recognition,&#x0201D;</article-title> in <source>2016 IEEE International Conference on Rebooting Computing (ICRC)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>1</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1109/ICRC.2016.7738683</pub-id></citation>
</ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rahimi</surname> <given-names>A.</given-names></name> <name><surname>Recht</surname> <given-names>B.</given-names></name></person-group> (<year>2007</year>). <article-title>&#x0201C;Random features for large-scale kernel machines,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source> 20.</citation>
</ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rahimi</surname> <given-names>A.</given-names></name> <name><surname>Tchouprina</surname> <given-names>A.</given-names></name> <name><surname>Kanerva</surname> <given-names>P.</given-names></name> <name><surname>Mill&#x000E1;n</surname> <given-names>J. D. R.</given-names></name> <name><surname>Rabaey</surname> <given-names>J. M.</given-names></name></person-group> (<year>2020</year>). <article-title>Hyperdimensional computing for blind and one-shot classification of EEG error-related potentials</article-title>. <source>Mobile Netw. Applic</source>. <volume>25</volume>, <fpage>958</fpage>&#x02013;<lpage>1969</lpage>. <pub-id pub-id-type="doi">10.1007/s11036-017-0942-6</pub-id></citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Roy</surname> <given-names>K.</given-names></name> <name><surname>Jaiswal</surname> <given-names>A.</given-names></name> <name><surname>Panda</surname> <given-names>P.</given-names></name></person-group> (<year>2019</year>). <article-title>Towards spike-based machine intelligence with neuromorphic computing</article-title>. <source>Nature</source> <volume>575</volume>, <fpage>607</fpage>&#x02013;<lpage>617</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-019-1677-2</pub-id><pub-id pub-id-type="pmid">31776490</pub-id></citation></ref>
<ref id="B36">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Rudin</surname> <given-names>W.</given-names></name></person-group> (<year>2017</year>). <source>Fourier Analysis on Groups</source>. <publisher-loc>Mineola, NY</publisher-loc>: <publisher-name>Courier Dover Publications</publisher-name>.</citation>
</ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Thomas</surname> <given-names>A.</given-names></name> <name><surname>Dasgupta</surname> <given-names>S.</given-names></name> <name><surname>Rosing</surname> <given-names>T.</given-names></name></person-group> (<year>2020</year>). <article-title>Theoretical foundations of hyperdimensional computing</article-title>. <source>arXiv preprint arXiv:2010.07426</source>.</citation>
</ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vanschoren</surname> <given-names>J.</given-names></name> <name><surname>van Rijn</surname> <given-names>J. N.</given-names></name> <name><surname>Bischl</surname> <given-names>B.</given-names></name> <name><surname>Torgo</surname> <given-names>L.</given-names></name></person-group> (<year>2013</year>). <article-title>Openml: networked science in machine learning</article-title>. <source>SIGKDD Explor</source>. <volume>15</volume>, <fpage>49</fpage>&#x02013;<lpage>60</lpage>. <pub-id pub-id-type="doi">10.1145/2641190.2641198</pub-id></citation>
</ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>R.</given-names></name> <name><surname>Jiao</surname> <given-names>X.</given-names></name> <name><surname>Hu</surname> <given-names>X. S.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Odhd: one-class brain-inspired hyperdimensional computing for outlier detection,&#x0201D;</article-title> in <source>Proceedings of the 59th ACM/IEEE Design Automation Conference</source> <fpage>43</fpage>&#x02013;<lpage>48</lpage>.</citation>
</ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zou</surname> <given-names>Z.</given-names></name> <name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Poduval</surname> <given-names>P.</given-names></name> <name><surname>Kim</surname> <given-names>Y.</given-names></name> <name><surname>Imani</surname> <given-names>M.</given-names></name> <name><surname>Sadredini</surname> <given-names>E.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>&#x0201C;Biohd: an efficient genome sequence search platform using hyperdimensional memorization,&#x0201D;</article-title> in <source>Proceedings of the 49th Annual International Symposium on Computer Architecture</source> 656&#x02013;669. <pub-id pub-id-type="doi">10.1145/3470496.3527422</pub-id></citation>
</ref>
</ref-list>
</back>
</article>