<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2025.1635932</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A brain-inspired memory transformation based differentiable neural computer for reasoning-based question answering</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Liang</surname> <given-names>Yao</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3079536/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Wang</surname> <given-names>Yuwei</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1049737/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Fang</surname> <given-names>Hongjian</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1099302/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhao</surname> <given-names>Feifei</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/534567/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Zeng</surname> <given-names>Yi</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/104116/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Brain-inspired Cognitive Intelligence Lab, Institute of Automation, Chinese Academy of Sciences</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>School of Artificial Intelligence, University of Chinese Academy of Sciences</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Center for Long-term Artificial Intelligence</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>School of Future Technology, University of Chinese Academy of Sciences</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff5"><sup>5</sup><institution>Key Laboratory of Brain Cognition and Brain-inspired Intelligence Technology, Chinese Academy of Sciences</institution>, <addr-line>Shanghai</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Kele Xu, National University of Defense Technology, China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Sanjay Singh, Manipal Institute of Technology, India</p>
<p>Gaojun Zhang, Tongji University, China</p>
<p>Jiazhen Xu, Central China Normal University, China</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Yi Zeng <email>yi.zeng&#x00040;ia.ac.cn</email></corresp>
<fn fn-type="equal" id="fn001"><p>&#x02020;These authors have contributed equally to this work and share first authorship</p></fn></author-notes>
<pub-date pub-type="epub">
<day>14</day>
<month>08</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>8</volume>
<elocation-id>1635932</elocation-id>
<history>
<date date-type="received">
<day>27</day>
<month>05</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>11</day>
<month>07</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Liang, Wang, Fang, Zhao and Zeng.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Liang, Wang, Fang, Zhao and Zeng</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Reasoning and question answering, as fundamental cognitive functions in humans, remain significant hurdles for artificial intelligence. While large language models (LLMs) have achieved notable success, integrating explicit memory with structured reasoning capabilities remains a persistent difficulty. The Differentiable Neural Computer (DNC) model, despite addressing these issues to some extent, still faces challenges such as algorithmic complexity, slow convergence, and limited robustness. Inspired by the brain&#x00027;s learning and memory mechanisms, this paper proposes a Memory Transformation based Differentiable Neural Computer (MT-DNC) model. The MT-DNC integrates two brain-inspired memory modules&#x02014;a working memory module inspired by the cognitive system that temporarily holds and processes task-relevant information, and a long-term memory module that stores frequently accessed and enduring information&#x02014;within the DNC framework, enabling the autonomous transformation of acquired experiences between these memory systems. This facilitates efficient knowledge extraction and enhances reasoning capabilities. Experimental results on the bAbI question answering task demonstrate that the proposed method outperforms existing Deep Neural Network (DNN) and DNC models, achieving faster convergence and superior performance. Ablation studies further confirm that the transformation of memory from working memory to long-term memory is critical for improving the robustness and stability of reasoning. This work offers new insights into incorporating brain-inspired memory mechanisms into dialogue and reasoning systems.</p></abstract>
<kwd-group>
<kwd>neural turing machine</kwd>
<kwd>memory-augmented networks</kwd>
<kwd>reasoning and question answering</kwd>
<kwd>working/long-term memory</kwd>
<kwd>differentiable neural computer</kwd>
</kwd-group>
<counts>
<fig-count count="3"/>
<table-count count="1"/>
<equation-count count="13"/>
<ref-count count="40"/>
<page-count count="0"/>
<word-count count="7283"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Machine Learning and Artificial Intelligence</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Reasoning and Question Answering (QA) are fundamental cognitive functions that are central to evaluating artificial intelligence systems. Despite the remarkable success of large language models (LLMs) (<xref ref-type="bibr" rid="B35">Touvron et al., 2023</xref>; <xref ref-type="bibr" rid="B9">Dubey et al., 2024</xref>; <xref ref-type="bibr" rid="B1">Achiam et al., 2023</xref>), challenges remain in developing methods that integrate explicit memory and structured reasoning capabilities. The Differentiable Neural Computer (DNC) model, proposed by (<xref ref-type="bibr" rid="B13">Graves et al. 2016</xref>), provides a feasible solution for studying reasoning and QA. DNC consists of a DNN-based computational controller and an external memory module, with which the neural network can interact (read and write). The memory module is responsible for representing and storing learned structures.</p>
<p>The DNC model has demonstrated good performance on various image reasoning and QA tasks (<xref ref-type="bibr" rid="B13">Graves et al., 2016</xref>; <xref ref-type="bibr" rid="B30">Rasekh and Safi-Esfahani, 2020</xref>). However, it faces several key challenges, including high algorithmic complexity, slow convergence speed, and a high average test error rate, all of which limit its further development and broader application. The BrsDNC model (<xref ref-type="bibr" rid="B10">Franke et al., 2018</xref>) improves the DNC model by introducing normalization and dropout, which have been shown to enhance robustness and scalability. The primary issues with current DNC models stem from restricted memory, which may lead to the loss of critical knowledge. As training time increases, the pressure on the memory module for reading and writing grows rapidly, thus limiting the model&#x00027;s training speed and performance. Besides, existing methods lack references from brain learning and memory mechanisms. Thus, there is still much room for improvement.</p>
<p>Memory in the brain encompasses both short-term and long-term memory, among others (<xref ref-type="bibr" rid="B5">Baddeley, 2007</xref>; <xref ref-type="bibr" rid="B27">Lee and Wilson, 2002</xref>; <xref ref-type="bibr" rid="B37">Winocur et al., 2010</xref>; <xref ref-type="bibr" rid="B28">Marshall and Born, 2007</xref>; <xref ref-type="bibr" rid="B18">Ji and Wilson, 2007</xref>). These types of memory play crucial roles in various cognitive functions, including learning, decision-making, and reasoning. Short-term memory has limited storage capacity and, therefore, cannot retain information indefinitely (<xref ref-type="bibr" rid="B8">Diamond, 2013</xref>). As a result, some memories are forgotten, while others that are repeatedly accessed are retained and transferred to long-term memory. Information can be stored in long-term memory for extended periods, continuously aiding learning and reasoning (<xref ref-type="bibr" rid="B2">Atkinson and Shiffrin, 1968</xref>). The collaboration and division of labor between working memory and long-term memory enable the brain to consolidate and apply acquired knowledge more efficiently, thereby enhancing the brain&#x00027;s capacity to perform multiple cognitive tasks (<xref ref-type="bibr" rid="B20">Kitamura et al., 2017</xref>). While short-term memory refers primarily to the brief retention of information, working memory further includes active manipulation and processing of information required for cognitive tasks, thus making it distinct and crucial for reasoning.</p>
<p>Inspired by the brain&#x00027;s learning and memory mechanisms, we propose a brain-inspired Memory Transformation based Differentiable Neural Computer (MT-DNC). Unlike the original DNC model, which has a single memory module, MT-DNC introduces two distinct memory modules: working memory and long-term memory. Working memory stores information directly relevant to the current task, while long-term memory holds more meaningful, enduring knowledge. These two memory modules are interconnected through a memory transformation algorithm. The core principles of the memory transformation algorithm are as follows: knowledge that is repeatedly accessed is transferred to long-term memory, while irrelevant information is discarded from working memory (<xref ref-type="bibr" rid="B40">Zhao et al., 2017</xref>; <xref ref-type="bibr" rid="B26">LeCun et al., 2015</xref>).</p>
<p>The innovations of our method are primarily reflected in the following aspects:</p>
<list list-type="order">
<list-item><p>Integration of working and long-term memory: MT-DNC introduces a novel architecture that explicitly combines working memory and long-term memory. This design enhances the model&#x00027;s ability to comprehensively store and utilize acquired knowledge, mimicking the human brain&#x00027;s memory system.</p></list-item>
<list-item><p>Brain-inspired memory transformation algorithm: A key contribution of MT-DNC is the development of a memory transformation algorithm inspired by biological memory mechanisms. This algorithm dynamically identifies and retains useful information by transferring it from working memory to long-term memory, while discarding irrelevant data, thereby optimizing memory efficiency.</p></list-item>
<list-item><p>Improved performance on reasoning tasks: Extensive experiments on the <italic>bAbI</italic> reasoning-based question-answering benchmark demonstrate that MT-DNC achieves superior accuracy and faster convergence compared to existing DNC-based methods. Moreover, the results highlight the crucial role of memory transformation in enhancing the model&#x00027;s stability and robustness during complex reasoning tasks.</p></list-item>
</list></sec>
<sec id="s2">
<title>2 Related work</title>
<p><bold>Neural Turing Machine (NTM):</bold> The core idea of NTM is to combine neural networks with external memory, thereby expanding the capabilities of neural networks and enabling interaction through an attention mechanism (<xref ref-type="bibr" rid="B12">Graves et al., 2014</xref>). To some extent, NTM can be compared to a Turing machine (<xref ref-type="bibr" rid="B38">Xiong et al., 2016</xref>; <xref ref-type="bibr" rid="B39">Zaremba and Sutskever, 2015</xref>), with experiments verifying its Turing completeness (<xref ref-type="bibr" rid="B34">Tao et al., 2021</xref>; <xref ref-type="bibr" rid="B39">Zaremba and Sutskever, 2015</xref>). The main advantage of NTM is its ability to handle complex tasks that require memory participation.</p>
<p><bold>Differentiable Neural Computer (DNC):</bold> DNC, which is considered an improved version of NTM, shares the same core idea of using external memory to enhance the ability of neural networks (<xref ref-type="bibr" rid="B13">Graves et al., 2016</xref>; <xref ref-type="bibr" rid="B31">Santoro et al., 2016</xref>; <xref ref-type="bibr" rid="B23">Lake et al., 2017</xref>). Compared to the original NTM, DNC introduces significant improvements in the addressing mechanism (<xref ref-type="bibr" rid="B14">Hassabis et al., 2017</xref>; <xref ref-type="bibr" rid="B6">Chan et al., 2018</xref>), removes the index shift operation, and better supports memory allocation and de-allocation functions. Additionally, DNC shows notable performance improvements over NTM.</p>
<p>Recent works have further enhanced the DNC architecture. (<xref ref-type="bibr" rid="B10">Franke et al. 2018</xref>) improved the model&#x00027;s performance by optimizing the memory module, increasing the bidirectional connections between memory modules, and introducing the layer normalization training method (Ba J. L. et al., <xref ref-type="bibr" rid="B4">2016</xref>). By refining the addressing and memory allocation processes, (<xref ref-type="bibr" rid="B7">Csord&#x000E1;s and Schmidhuber 2019</xref>) achieved better accuracy on the bAbI task. (<xref ref-type="bibr" rid="B30">Rasekh and Safi-Esfahani 2020</xref>) integrated the NeuroEvolution algorithm into the DNC framework, demonstrating faster encoding speed in various cognitive tasks, leading to improved model performance.</p>
<p>To summarize, none of these approaches fully address the issues of low accuracy and slow convergence associated with DNC&#x00027;s limited external memory. This paper draws inspiration from the brain&#x00027;s learning and memory mechanisms and proposes the MT-DNC model, which integrates two coordinated memory modules: working memory and long-term memory (<xref ref-type="bibr" rid="B32">Seo et al., 2016</xref>; Ba J. et al., <xref ref-type="bibr" rid="B3">2016</xref>; <xref ref-type="bibr" rid="B24">Le et al., 2019</xref>, <xref ref-type="bibr" rid="B25">2020</xref>). The proposed model improves both accuracy and convergence speed, offering superior performance compared to existing DNC-based models.</p></sec>
<sec id="s3">
<title>3 Method</title>
<p>In this section, we provide a comprehensive introduction to the MT-DNC model. MT-DNC extends the memory module of the DNC by incorporating both a working memory module and a long-term memory module. Inspired by the brain&#x00027;s learning and memory mechanisms, MT-DNC introduces a dual-memory architecture that consists of both working memory and long-term memory. This architecture enables the model to manage and store information more effectively, thereby enhancing its reasoning and knowledge retention capabilities. The core innovation lies in a dynamic memory transformation mechanism that selectively transfers frequently accessed or meaningful information from working memory to long-term memory, enabling the model to maintain a compact yet informative working memory.</p>
<p>In the MT-DNC architecture, working memory (or short-term memory) rapidly processes and updates information needed immediately, while the long-term memory persistently retains valuable knowledge, with the memory transformation mechanism dynamically managing information transfer between these memory modules to enhance reasoning efficiency.</p>
<p>The overall framework of MT-DNC consists of three layers: the controller layer, memory layer, and linear layer, as shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. The controller layer is responsible for encoding and processing both the input data and the output from the previous time step of the controller layer and the memory layer, learning temporal patterns from the training data, and transmitting the results to both the memory and linear layers. The memory layer is responsible for storing the controller&#x00027;s output and extracting useful information through a series of storage and transformation mechanisms. This layer also incorporates memory transformation between the working memory and long-term memory modules, enabling the MT-DNC model to exhibit strong memory and reasoning capabilities. The linear layer combines the outputs from the controller and memory layers, and produces the final prediction result via a linear transformation.</p>
<fig position="float" id="F1">
<label>Figure 1</label>
<caption><p>Overall architecture of MT-DNC.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1635932-g0001.tif">
<alt-text>Diagram illustrating a neural network architecture with three main layers Controller Layer, Memory Layer, and Linear Layer. Inputs feed into the Controller Layer, which interacts with the Memory Layer through signal generation, involving write keys, vectors, and erase vector operations. The Memory Layer includes a Memory Manager that coordinates working memory and long-term memory. Back vectors process read vectors. Outputs are produced by the Linear Layer. Arrows indicate data flow between layers and components.</alt-text>
</graphic>
</fig>
<sec>
<title>3.1 Controller layer</title>
<p>The controller layer combines the original input data <inline-formula><mml:math id="M1"><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> with the output of the memory layer from the previous time step, <inline-formula><mml:math id="M2"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mi>R</mml:mi><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, as well as the output of the controller layer from the previous time step, after undergoing Dropout processing. After performing a Long Short-Term Memory (LSTM) operation and applying layer normalization (<xref ref-type="bibr" rid="B21">Klambauer et al., 2017</xref>; <xref ref-type="bibr" rid="B10">Franke et al., 2018</xref>), the resulting output <inline-formula><mml:math id="M3"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> is transmitted to the memory layer. As shown in <xref ref-type="disp-formula" rid="E1">Equation 1</xref>:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M4"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:msubsup><mml:mi>O</mml:mi><mml:mi>t</mml:mi><mml:mi>c</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:mtext>LayerNorm</mml:mtext><mml:mo>&#x02212;</mml:mo><mml:mtext>LSTM</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>&#x02295;</mml:mo><mml:msubsup><mml:mi>O</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>m</mml:mi></mml:msubsup><mml:mo>&#x02295;</mml:mo><mml:mtext>Dropout</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mi>O</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>c</mml:mi></mml:msubsup><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>c</mml:mi></mml:mstyle><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>;</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:msubsup><mml:mi mathvariant='script'>W</mml:mi><mml:mi>t</mml:mi><mml:mi>c</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>b</mml:mi><mml:mi>t</mml:mi><mml:mi>c</mml:mi></mml:msubsup><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M6"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>C</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> denotes the output at time step <italic>t</italic>. The term <bold>c</bold><sub><italic>t</italic>&#x02212;1</sub> represents the cell state from the previous time step, and <inline-formula><mml:math id="M7"><mml:mrow><mml:msubsup><mml:mrow><mml:mi mathvariant="script">W</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>2</mml:mn><mml:mi>R</mml:mi><mml:mi>W</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>C</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000D7;</mml:mo><mml:mi>C</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> is the weight matrix that maps the input to the gates. Additionally, <inline-formula><mml:math id="M8"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>C</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> is the bias vector associated with the input to the gates, and &#x02295; denotes the concatenation of vectors.</p>
<p>Here, <italic>X</italic> represents the dimension of the input data, <italic>C</italic> represents the output dimension of the controller layer, and <italic>W</italic> represents the width of the memory region.</p>
</sec>
<sec>
<title>3.2 Memory layer</title>
<p>The memory layer consists of the working memory module (functionally analogous to working memory in human cognition, temporarily storing and actively processing task-relevant information), the long-term memory module (storing enduring and frequently accessed knowledge), and the memory transformation mechanism. The working memory module stores the most recent interaction data from the controller layer, while the long-term memory holds frequently used information of high importance that may eventually be discarded by the working memory. Both the working memory and long-term memory require dynamic update and extraction rules to continuously replace stored information. The memory transformation mechanism selectively transfers data from working memory to long-term memory for processing. Finally, the memory layer combines the outputs from the controller layer, working memory, and long-term memory to make decisions.</p>
<sec>
<title>3.2.1 Working memory module</title>
<p>The working memory module is functionally designed to store interactive information from the controller layer&#x00027;s output in real time, updating and extracting relevant information based on the controller layer&#x00027;s output. Due to storage limitations, we draw inspiration from the memory update and decay mechanisms in the human brain, replacing information that is similar to the current interaction data (<inline-formula><mml:math id="M9"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>). Additionally, information that has already been extracted or used is more likely to be replaced in order to retain as much novel information as possible.</p>
<p>The read, write, and gating signals within the memory region are generated from <inline-formula><mml:math id="M10"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> through a linear transformation. Let <inline-formula><mml:math id="M11"><mml:mrow><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mi>R</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>6</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>W</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>6</mml:mn><mml:mo>&#x0002B;</mml:mo><mml:mn>4</mml:mn><mml:mi>R</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> represent the signal vector at time step <italic>t</italic>, derived via layer normalization, as shown in <xref ref-type="disp-formula" rid="E2">Equation 2</xref>:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M12"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext>LayerNormalization</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x000B7;</mml:mo><mml:msubsup><mml:mrow><mml:mi mathvariant="script">W</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msubsup><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M13"><mml:mrow><mml:msubsup><mml:mrow><mml:mi mathvariant="script">W</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mi>R</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>6</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>W</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>6</mml:mn><mml:mo>&#x0002B;</mml:mo><mml:mn>4</mml:mn><mml:mi>R</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> is the weight matrix and <inline-formula><mml:math id="M14"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mi>R</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>6</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>W</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>6</mml:mn><mml:mo>&#x0002B;</mml:mo><mml:mn>4</mml:mn><mml:mi>R</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> is the bias vector. The dimension of <italic>S</italic><sub><italic>t</italic></sub> is carefully designed based on the operational needs of both working and long-term memory modules, involving signals for writing, reading, erasing, and gating controls. Specifically, (2<italic>R</italic>&#x0002B;6)<italic>W</italic> represents memory signals corresponding to multiple read/write operations across working and long-term memories, while the additional terms 6 and 4<italic>R</italic> account for scalar gates and strengths. A comprehensive step-by-step derivation is provided in <xref ref-type="app" rid="A1">Appendix A</xref>.</p>
<p>This normalized signal vector is systematically partitioned into several distinct components, each corresponding to specific memory regions and operational functionalities, ensuring that the total length of all variables matches the dimension of <italic>S</italic><sub><italic>t</italic></sub>.</p>
<p>Initially, the first <italic>W</italic> elements of <italic>S</italic><sub><italic>t</italic></sub> are designated as the write query signal for the working memory region, denoted by <inline-formula><mml:math id="M15"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, while the subsequent <italic>W</italic> elements serve as the write query signal for the long-term memory region, denoted by <inline-formula><mml:math id="M16"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>. Following these, the next two elements are processed through the <monospace>oneplus</monospace> activation function to yield the write scaling factors <inline-formula><mml:math id="M17"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:mi>&#x0211D;</mml:mi></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M18"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:mi>&#x0211D;</mml:mi></mml:mrow></mml:math></inline-formula> for the working and long-term memory regions, respectively. The <monospace>oneplus</monospace> function is defined as:</p>
<disp-formula id="E3"><mml:math id="M19"><mml:mrow><mml:mtext class="texttt" mathvariant="monospace">oneplus</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x0002B;</mml:mo><mml:mtext class="textrm" mathvariant="normal">softplus</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x0002B;</mml:mo><mml:mo class="qopname">ln</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>This function ensures that the scaling factors are strictly positive, facilitating stable and controlled scaling during the write operations.</p>
<p>Subsequently, the next 2<italic>W</italic> elements of <italic>S</italic><sub><italic>t</italic></sub> are passed through the sigmoid activation function to generate the erase signals <inline-formula><mml:math id="M20"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M21"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, which facilitate the controlled removal of information within the working and long-term memory regions, respectively. The following 2<italic>W</italic> elements are directly extracted to form the write signals <inline-formula><mml:math id="M22"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M23"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, enabling the storage of new information.</p>
<p>To regulate weight allocation and the strength of write operations, the subsequent four elements are processed through the sigmoid function to derive the gating scalars <inline-formula><mml:math id="M24"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:mi>&#x0211D;</mml:mi></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="M25"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:mi>&#x0211D;</mml:mi></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="M26"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:mi>&#x0211D;</mml:mi></mml:mrow></mml:math></inline-formula>, and <inline-formula><mml:math id="M27"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:mi>&#x0211D;</mml:mi></mml:mrow></mml:math></inline-formula>. These gating scalars modulate the write operations within both the working and long-term memory regions effectively.</p>
<p>For multi-head read operations, the signal vector is further partitioned into components corresponding to each of the <italic>R</italic> read heads. Specifically, the read query signals <inline-formula><mml:math id="M28"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>R</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M29"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>R</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> are extracted for the working and long-term memory regions, respectively. The corresponding read scaling factors <inline-formula><mml:math id="M30"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M31"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> are obtained by applying the <monospace>oneplus</monospace> function to the relevant segments of <italic>S</italic><sub><italic>t</italic></sub>. Additionally, the free gating vectors <inline-formula><mml:math id="M32"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M33"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> are computed using the sigmoid function, providing flexible control over information retrieval across all read heads.</p>
<p>Here, <italic>W</italic> represents the width of the memory region, and <italic>R</italic> specifies the number of read heads. The regions <italic>wk</italic> and <italic>lt</italic> refer to the working and long-term memory, respectively, while <italic>t</italic> denotes the current time step in the signal processing sequence. The total length of all these variables collectively equals the dimension of <italic>S</italic><sub><italic>t</italic></sub>, which is (2<italic>R</italic>&#x0002B;6)<italic>W</italic>&#x0002B;6&#x0002B;4<italic>R</italic>. This meticulous segmentation of <italic>S</italic><sub><italic>t</italic></sub> into dedicated variables, each with explicitly defined dimensionalities, ensures efficient and optimized storage and retrieval processes across both memory regions. Consequently, this enhances the overall functionality and performance of the working memory module by enabling precise control and manipulation of information within the system.</p>
<p><bold>Working Memory Updating Algorithm</bold>. The updating of the working memory is based on the following principles:</p>
<list list-type="order">
<list-item><p>Delete memory slots with lower usage frequency or longer recency intervals. Specifically, items with the lowest usage value, tracked by the usage vector <inline-formula><mml:math id="M34"><mml:msubsup><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>, are prioritized for deletion. The usage vector is updated at each time step based on previous read and write weights, which progressively reduces the usage value of slots that have not been accessed or updated recently.</p></list-item>
<list-item><p>Delete items after extraction, which corresponds to actively setting low retention values using the free gates (<inline-formula><mml:math id="M35"><mml:msubsup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>), effectively marking them for replacement.</p></list-item>
<list-item><p>Delete memory items whose content is highly similar to newly stored information. The similarity is measured by cosine similarity in content-based addressing.</p></list-item>
<list-item><p>Retain recently updated novel items, identified as slots with recent write operations and relatively higher usage values in the usage vector.</p></list-item>
</list>
<p>Based on these principles, we update the working memory in real-time according to the dynamic addressing algorithm in <xref ref-type="disp-formula" rid="E3">Equation 3</xref> (<xref ref-type="bibr" rid="B13">Graves et al., 2016</xref>; <xref ref-type="bibr" rid="B17">Hsin, 2016</xref>).</p>
<disp-formula id="E4"><label>(3)</label><mml:math id="M36"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:msubsup><mml:mi>&#x003C8;</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x0220F;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>R</mml:mi></mml:munderover><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:msubsup><mml:mi>f</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:msubsup><mml:mi>C</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mstyle><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:msubsup><mml:mi>U</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mi>U</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:msubsup><mml:mi mathvariant='script'>W</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02212;</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mi>U</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x000B0;</mml:mo><mml:msubsup><mml:mi mathvariant='script'>W</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x000B0;</mml:mo><mml:msubsup><mml:mi>&#x003C8;</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:msubsup><mml:mi>&#x003D5;</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mtext>SortIndiceAscending</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mi>U</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msubsup><mml:mi>A</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy='false'>[</mml:mo><mml:msubsup><mml:mi>&#x003D5;</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy='false'>[</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy='false'>]</mml:mo><mml:mo stretchy='false'>]</mml:mo><mml:mo>=</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:msubsup><mml:mi>U</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy='false'>[</mml:mo><mml:msubsup><mml:mi>&#x003D5;</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy='false'>[</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy='false'>]</mml:mo><mml:mo stretchy='false'>]</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x0220F;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:munderover><mml:mrow><mml:msubsup><mml:mi>U</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:mstyle><mml:mo stretchy='false'>[</mml:mo><mml:msubsup><mml:mi>&#x003D5;</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy='false'>[</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy='false'>]</mml:mo><mml:mo stretchy='false'>]</mml:mo><mml:mo stretchy='false'>]</mml:mo><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M37"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> is the result of scaling and accumulating the read weight matrix <inline-formula><mml:math id="M38"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> from the previous time step using the <inline-formula><mml:math id="M39"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> gated vector. The <inline-formula><mml:math id="M40"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> tensor is the index tensor, sorted in ascending order by the memory region management tensor <inline-formula><mml:math id="M41"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, where <italic>N</italic> represents the length of the memory region. Additionally, <inline-formula><mml:math id="M42"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> represents the write weight of the working memory region based on dynamic addressing.</p>
<p>Specifically, in <xref ref-type="disp-formula" rid="E3">Equation 3</xref>, the tensor <inline-formula><mml:math id="M43"><mml:msubsup><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> precisely tracks the usage frequency and recency of each memory slot. A low value in <inline-formula><mml:math id="M44"><mml:msubsup><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> directly indicates infrequent access or prolonged non-usage. The free gate vectors (<inline-formula><mml:math id="M45"><mml:msubsup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>) from multiple read heads further modulate the retention values of memory slots, explicitly controlling the deletion of recently extracted items. Consequently, memory slots with persistently low <inline-formula><mml:math id="M46"><mml:msubsup><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> values, resulting from limited read/write activities over multiple consecutive time steps, are considered to have not been used for a &#x0201C;long time&#x0201D; and thus are candidates for deletion.</p>
<p>The method for calculating write weights based on content addressing in the working memory region is presented in <xref ref-type="disp-formula" rid="E5">Equation 4</xref> (<xref ref-type="bibr" rid="B13">Graves et al., 2016</xref>; <xref ref-type="bibr" rid="B17">Hsin, 2016</xref>):</p>
<disp-formula id="E5"><label>(4)</label><mml:math id="M47"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>d</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mo>&#x02211;</mml:mo><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>d</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle displaystyle="true"><mml:msubsup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mstyle><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M48"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="M49"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:mi>&#x0211D;</mml:mi></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="M50"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, and <inline-formula><mml:math id="M51"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> represent the working memory region, and <inline-formula><mml:math id="M52"><mml:mrow><mml:mi>d</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>u</mml:mi><mml:mo>,</mml:mo><mml:mi>v</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>u</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mi>u</mml:mi><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:mi>v</mml:mi><mml:mo>|</mml:mo></mml:mrow></mml:mfrac></mml:mrow></mml:math></inline-formula>. Here, <italic>N</italic> represents the length of the memory region, and <italic>W</italic> represents the width of the memory region.</p>
<p>The write algorithm for the working memory region is presented in <xref ref-type="disp-formula" rid="E6">Equation 5</xref> (<xref ref-type="bibr" rid="B13">Graves et al., 2016</xref>; <xref ref-type="bibr" rid="B17">Hsin, 2016</xref>):</p>
<disp-formula id="E6"><label>(5)</label><mml:math id="M53"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi mathvariant="script">W</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:msubsup><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>-</mml:mo><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x000B0;</mml:mo><mml:msubsup><mml:mrow><mml:mi mathvariant="script">W</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msubsup><mml:mrow><mml:mi mathvariant="script">W</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M54"><mml:mrow><mml:msubsup><mml:mrow><mml:mi mathvariant="script">W</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> represents the final write weight of the working memory region, and <inline-formula><mml:math id="M55"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> denotes the write weight allocation gate scalar, which controls the allocation proportion of the two addressing modes in the final write. The gating scalar <inline-formula><mml:math id="M56"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> serves to protect the data in the memory region, preserving its relative stability and preventing it from being overwhelmed by unimportant, redundant, or irrelevant information.</p>
<p><bold>Working Memory Extraction Algorithm</bold>. In the extraction of working memory, the information most relevant to the current interactive read query signal <inline-formula><mml:math id="M57"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> is retrieved. The extraction weighting algorithm is defined by <xref ref-type="disp-formula" rid="E7">Equation 6</xref> as follows (<xref ref-type="bibr" rid="B13">Graves et al., 2016</xref>; <xref ref-type="bibr" rid="B17">Hsin, 2016</xref>):</p>
<disp-formula id="E7"><label>(6)</label><mml:math id="M58"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>d</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mo>&#x02211;</mml:mo><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>d</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle displaystyle="true"><mml:msubsup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mstyle><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M59"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>R</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>N</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="M60"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>R</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, and <inline-formula><mml:math id="M61"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, with <italic>R</italic> representing the total number of read operations, and <italic>i</italic> indicating the specific label.</p>
<p>The information extraction algorithm within the working memory region is defined by <xref ref-type="disp-formula" rid="E8">Equation 7</xref> as follows:</p>
<disp-formula id="E8"><label>(7)</label><mml:math id="M62"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M63"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>R</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>.</p></sec>
<sec>
<title>3.2.2 Memory transformation mechanism</title>
<p>The DNC-based model (<xref ref-type="bibr" rid="B13">Graves et al., 2016</xref>; <xref ref-type="bibr" rid="B10">Franke et al., 2018</xref>) directly maps the output of the working memory (<inline-formula><mml:math id="M64"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>) to a linear layer. However, since the items that have been used are deleted from working memory, this leads to the loss of important information, which in turn affects both performance and robustness. We propose a memory transformation algorithm that transfers information extracted from the working memory into the long-term memory, compensating for information loss due to frequent updates and deletions in the working memory.</p>
<p>The algorithm for updating and extracting information in long-term memory is similar to that in working memory. The only difference is that the input in working memory originates from the controller layer, whereas the input in long-term memory originates from the working memory module. The update formula for the long-term memory region is given in <xref ref-type="disp-formula" rid="E9">Equation 8</xref>:</p>
<disp-formula id="E9"><label>(8)</label><mml:math id="M65"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi mathvariant="script">W</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:msubsup><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>B</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x0220F;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>-</mml:mo><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x000B0;</mml:mo><mml:msubsup><mml:mrow><mml:mi mathvariant="script">W</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msubsup><mml:mrow><mml:mi mathvariant="script">W</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>B</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr></mml:mtr></mml:mtable></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M66"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="M67"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>B</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, and <inline-formula><mml:math id="M68"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> represents the long-term memory write weight allocation gate scalar, which controls the allocation proportion of the two addressing modes in the final write.</p>
<p>Information extraction from the memory layer integrates information from both the working memory region, <inline-formula><mml:math id="M69"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>, and the long-term memory region, <inline-formula><mml:math id="M70"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>. This calculation is given by the following equation in <xref ref-type="disp-formula" rid="E10">Equation 9</xref>:</p>
<disp-formula id="E10"><label>(9)</label><mml:math id="M71"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mi mathvariant="script">R</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02295;</mml:mo><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M72"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mi>R</mml:mi><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, and R(&#x000B7;) represents the reshaping operation applied to the concatenated tensor <inline-formula><mml:math id="M73"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02295;</mml:mo><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>, transforming it into a vector of length 2<italic>RW</italic>.</p>
</sec>
</sec>
<sec>
<title>3.3 Linear layer</title>
<p>The output of the linear layer, &#x00177;<sub><italic>t</italic></sub>, is determined by the output of the controller layer, <inline-formula><mml:math id="M74"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>, after Dropout processing (<xref ref-type="bibr" rid="B10">Franke et al., 2018</xref>; <xref ref-type="bibr" rid="B11">Gal and Ghahramani, 2016</xref>; <xref ref-type="bibr" rid="B33">Srivastava et al., 2014</xref>), as well as the output of the memory layer, <inline-formula><mml:math id="M75"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>, given by <xref ref-type="disp-formula" rid="E11">Equation 10</xref>:</p>
<disp-formula id="E11"><label>(10)</label><mml:math id="M76"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x00177;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext>Softmax</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02295;</mml:mo><mml:mtext>Dropout</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000B7;</mml:mo><mml:msubsup><mml:mrow><mml:mi mathvariant="script">W</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msubsup><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M77"><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x00177;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>Y</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="M78"><mml:mrow><mml:msubsup><mml:mrow><mml:mi mathvariant="script">W</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mi>R</mml:mi><mml:mi>W</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>C</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000D7;</mml:mo><mml:mi>Y</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> is the output weight matrix, and <inline-formula><mml:math id="M79"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>Y</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> is the bias vector.</p>
<p>The detailed procedure of our MT-DNC model is shown in <xref ref-type="table" rid="T2">Algorithm 1</xref>.</p>
<table-wrap position="float" id="T2">
<label>Algorithm 1</label>
<caption><p>Execution algorithm for MT-DNC.</p></caption>
<graphic xlink:href="frai-08-1635932-i0001.tif"/>
</table-wrap>
</sec>
</sec>
<sec id="s4">
<title>4 Experiments</title>
<sec>
<title>4.1 The bAbI task</title>
<p>The bAbI<xref ref-type="fn" rid="fn0001"><sup>1</sup></xref> is a reasoning-based text question-and-answer task (<xref ref-type="bibr" rid="B36">Weston et al., 2015</xref>; <xref ref-type="bibr" rid="B22">Kumar et al., 2016</xref>). We use the en-10k dataset for experimentation, which contains 20 sub-tasks. Each subtask contains numerous stories, with each story consisting of supporting facts, multiple questions, and their corresponding answers. The correct answers rely on one or more supporting facts. A joint training approach is employed to evaluate the text comprehension and reasoning ability of the MT-DNC model. Unlike other previous works, our method uses end-to-end training without any pre-processing of the bAbI dataset itself.</p>
</sec>
<sec>
<title>4.2 Training details</title>
<p>The bAbI question-and-answer task, comprising 20 sub-tasks, is combined into a single training session. A training sample is generated for each sub-task in the dataset, based on different stories. The detailed generation process is as follows:</p>
<list list-type="order">
<list-item><p>The text sequence training samples are processed by removing digits, converting words to lowercase, removing line breaks, etc.</p></list-item>
<list-item><p>The text sequence training samples are split into lists of word sequences (including 3 punctuation marks).</p></list-item>
<list-item><p>The &#x0201C;answer words&#x0201D; in the list are replaced with &#x0201C;-&#x0201D;, and the list is then encoded into word vectors using a one-hot word vector processor. The length of the list corresponds to the length of the largest text sequence in the current batch, and shorter texts are padded with &#x0201C;0&#x0201D;. A word in the list is represented as <inline-formula><mml:math id="M85"><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, where <italic>X</italic> is the length of the word vector, with a value of 159.</p></list-item>
<list-item><p>All training input samples and target samples are combined to form the training sample list.</p></list-item>
<list-item><p>10% of the data in the training sample list is used as the validation dataset.</p></list-item>
<list-item><p>The MT-DNC model is trained for 300 epochs, with validation and testing after each epoch.</p></list-item>
</list>
<p>The total number of parameters in the model is 1,267,337, and the batch size is 32. The number of control layer nodes is 172, corresponding to the output dimension <italic>C</italic> of the control layer. Both memory regions have a length of 128 (i.e., dimension <italic>N</italic>) and a width of 64 (i.e., dimension <italic>W</italic>), with 4 read heads (i.e., <italic>R</italic>), 1 write head, and a dropout rate of 0.9. The learning rate is 0.0003, and the momentum value of the Rmsprop optimizer is 0.9 (<xref ref-type="bibr" rid="B19">Kingma and Ba, 2014</xref>). The gradient clipping value is set to 10.</p>
</sec>
<sec>
<title>4.3 Experimental results</title>
<p>To verify the effectiveness of the proposed MT-DNC model, we conducted comparison experiments with DNC, EntNet (<xref ref-type="bibr" rid="B15">Henaff et al., 2016</xref>), LSTM (<xref ref-type="bibr" rid="B16">Hochreiter and Schmidhuber, 1997</xref>), SDNC (<xref ref-type="bibr" rid="B29">Rae et al., 2016</xref>), BrsDNC (<xref ref-type="bibr" rid="B10">Franke et al., 2018</xref>), and other models on the bAbI question-and-answer task. Additionally, we evaluated the MT-DNC-DI model (a variant of our MT-DNC model without the memory transformation mechanism, where &#x0201C;DI&#x0201D; stands for Direct Independence) to assess the impact of the memory transformation algorithm on model performance. The MT-DNC-DI model employs independent memory modules, with separate regions for working memory and long-term memory, both of which receive input directly from the controller layer. <xref ref-type="table" rid="T1">Table 1</xref> shows the average word error rate (WER) of different models under different initialized parameters.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>The average word error rate (WER) of different models on bAbI task (<xref ref-type="bibr" rid="B15">Henaff et al., 2016</xref>; <xref ref-type="bibr" rid="B16">Hochreiter and Schmidhuber, 1997</xref>; <xref ref-type="bibr" rid="B29">Rae et al., 2016</xref>; <xref ref-type="bibr" rid="B10">Franke et al., 2018</xref>).</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#8f9496;color:#ffffff">
<th valign="top" align="left"><bold>Task</bold></th>
<th valign="top" align="center"><bold>DNC</bold></th>
<th valign="top" align="center"><bold>EntNet</bold></th>
<th valign="top" align="center"><bold>LSTM</bold></th>
<th valign="top" align="center"><bold>SDNC</bold></th>
<th valign="top" align="center"><bold>BrsDNC</bold></th>
<th valign="top" align="center"><bold>MT-DNC-DI</bold></th>
<th valign="top" align="center"><bold>MT-DNC</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1: 1 supporting fact</td>
<td valign="top" align="center">9.0 &#x000B1; 12.6</td>
<td valign="top" align="center">0.0 &#x000B1; 0.1</td>
<td valign="top" align="center">28.4 &#x000B1; 1.5</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center">0.1 &#x000B1; 0.1</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center"><bold>0.0</bold> <bold>&#x000B1;0.0</bold></td>
</tr>
<tr>
<td valign="top" align="left">2: 2 supporting facts</td>
<td valign="top" align="center">39.2 &#x000B1; 20.5</td>
<td valign="top" align="center">15.3 &#x000B1; 15.7</td>
<td valign="top" align="center">56.0 &#x000B1; 1.5</td>
<td valign="top" align="center">7.1 &#x000B1; 14.6</td>
<td valign="top" align="center">0.8 &#x000B1; 0.2</td>
<td valign="top" align="center">0.3 &#x000B1; 0.2</td>
<td valign="top" align="center"><bold>0.4</bold> <bold>&#x000B1;0.3</bold></td>
</tr>
<tr>
<td valign="top" align="left">3: 3 supporting facts</td>
<td valign="top" align="center">39.6 &#x000B1; 16.4</td>
<td valign="top" align="center">29.3 &#x000B1; 26.3</td>
<td valign="top" align="center">51.3 &#x000B1; 1.4</td>
<td valign="top" align="center">9.4 &#x000B1; 16.7</td>
<td valign="top" align="center">2.4 &#x000B1; 0.6</td>
<td valign="top" align="center">2.8 &#x000B1; 0.8</td>
<td valign="top" align="center">2.7 &#x000B1; 0.8</td>
</tr>
<tr>
<td valign="top" align="left">4: 2 argument relations</td>
<td valign="top" align="center">0.4 &#x000B1; 0.7</td>
<td valign="top" align="center">0.1 &#x000B1; 0.1</td>
<td valign="top" align="center">0.8 &#x000B1; 0.5</td>
<td valign="top" align="center">0.1 &#x000B1; 0.1</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center"><bold>0.0</bold> <bold>&#x000B1;0.0</bold></td>
</tr>
<tr>
<td valign="top" align="left">5: 3 argument relations</td>
<td valign="top" align="center">1.5 &#x000B1; 1.0</td>
<td valign="top" align="center">0.4 &#x000B1; 0.3</td>
<td valign="top" align="center">3.2 &#x000B1; 0.5</td>
<td valign="top" align="center">0.9 &#x000B1; 0.3</td>
<td valign="top" align="center">0.7 &#x000B1; 0.1</td>
<td valign="top" align="center">0.6 &#x000B1; 0.3</td>
<td valign="top" align="center"><bold>0.5</bold> <bold>&#x000B1;0.1</bold></td>
</tr>
<tr>
<td valign="top" align="left">6: yes/no questions</td>
<td valign="top" align="center">6.9 &#x000B1; 7.5</td>
<td valign="top" align="center">0.6 &#x000B1; 0.8</td>
<td valign="top" align="center">15.2 &#x000B1; 1.5</td>
<td valign="top" align="center">0.1 &#x000B1; 0.2</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center"><bold>0.0</bold> <bold>&#x000B1;0.0</bold></td>
</tr>
<tr>
<td valign="top" align="left">7: counting</td>
<td valign="top" align="center">9.8 &#x000B1; 7.0</td>
<td valign="top" align="center">1.8 &#x000B1; 1.1</td>
<td valign="top" align="center">16.4 &#x000B1; 1.4</td>
<td valign="top" align="center">1.6 &#x000B1; 0.9</td>
<td valign="top" align="center">1.0 &#x000B1; 0.5</td>
<td valign="top" align="center">0.6 &#x000B1; 0.3</td>
<td valign="top" align="center"><bold>0.6</bold> <bold>&#x000B1;0.2</bold></td>
</tr>
<tr>
<td valign="top" align="left">8: lists/sets</td>
<td valign="top" align="center">5.5 &#x000B1; 5.9</td>
<td valign="top" align="center">1.5 &#x000B1; 1.2</td>
<td valign="top" align="center">17.7 &#x000B1; 1.2</td>
<td valign="top" align="center">0.5 &#x000B1; 0.4</td>
<td valign="top" align="center">0.5 &#x000B1; 0.3</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center"><bold>0.1</bold> <bold>&#x000B1;0.1</bold></td>
</tr>
<tr>
<td valign="top" align="left">9: simple negation</td>
<td valign="top" align="center">7.7 &#x000B1; 8.3</td>
<td valign="top" align="center">0.0 &#x000B1; 0.1</td>
<td valign="top" align="center">15.4 &#x000B1; 1.5</td>
<td valign="top" align="center">0.0 &#x000B1; 0.1</td>
<td valign="top" align="center">0.1 &#x000B1; 0.2</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center"><bold>0.0</bold> <bold>&#x000B1;0.0</bold></td>
</tr>
<tr>
<td valign="top" align="left">10: indefinite knowledge</td>
<td valign="top" align="center">9.6 &#x000B1; 11.4</td>
<td valign="top" align="center">0.1 &#x000B1; 0.2</td>
<td valign="top" align="center">28.7 &#x000B1; 1.7</td>
<td valign="top" align="center">0.3 &#x000B1; 0.2</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center"><bold>0.0</bold> <bold>&#x000B1;0.0</bold></td>
</tr>
<tr>
<td valign="top" align="left">11: basic coreference</td>
<td valign="top" align="center">3.3 &#x000B1; 5.7</td>
<td valign="top" align="center">0.2 &#x000B1; 0.2</td>
<td valign="top" align="center">12.2 &#x000B1; 3.5</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center"><bold>0.0</bold> <bold>&#x000B1;0.0</bold></td>
</tr>
<tr>
<td valign="top" align="left">12: conjunction</td>
<td valign="top" align="center">5 &#x000B1; 6.3</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center">5.4 &#x000B1; 0.6</td>
<td valign="top" align="center">0.2 &#x000B1; 0.3</td>
<td valign="top" align="center">0.0 &#x000B1; 0.1</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center"><bold>0.0</bold> <bold>&#x000B1;0.0</bold></td>
</tr>
<tr>
<td valign="top" align="left">13: compound coreference</td>
<td valign="top" align="center">3.1 &#x000B1; 3.6</td>
<td valign="top" align="center">0.0 &#x000B1; 0.1</td>
<td valign="top" align="center">7.2 &#x000B1; 2.3</td>
<td valign="top" align="center">0.1 &#x000B1; 0.1</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center"><bold>0.0</bold> <bold>&#x000B1;0.0</bold></td>
</tr>
<tr>
<td valign="top" align="left">14: time reasoning</td>
<td valign="top" align="center">11 &#x000B1; 7.5</td>
<td valign="top" align="center">7.3 &#x000B1; 4.5</td>
<td valign="top" align="center">55.9 &#x000B1; 1.2</td>
<td valign="top" align="center">5.6 &#x000B1; 2.9</td>
<td valign="top" align="center">0.8 &#x000B1; 0.7</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center"><bold>0.0</bold> <bold>&#x000B1;0.0</bold></td>
</tr>
<tr>
<td valign="top" align="left">15: basic deduction</td>
<td valign="top" align="center">27.2 &#x000B1; 20.1</td>
<td valign="top" align="center">3.6 &#x000B1; 8.1</td>
<td valign="top" align="center">47.0 &#x000B1; 1.7</td>
<td valign="top" align="center">3.6 &#x000B1; 10.3</td>
<td valign="top" align="center">0.1 &#x000B1; 0.1</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center"><bold>0.0</bold> <bold>&#x000B1;0.0</bold></td>
</tr>
<tr>
<td valign="top" align="left">16: basic induction</td>
<td valign="top" align="center">53.6 &#x000B1; 1.9</td>
<td valign="top" align="center">53.3 &#x000B1; 1.2</td>
<td valign="top" align="center">53.3 &#x000B1; 1.3</td>
<td valign="top" align="center">53.0 &#x000B1; 1.3</td>
<td valign="top" align="center">52.6 &#x000B1; 1.6</td>
<td valign="top" align="center">49.1 &#x000B1; 0.9</td>
<td valign="top" align="center"><bold>38.8</bold> <bold>&#x000B1;11.1</bold></td>
</tr>
<tr>
<td valign="top" align="left">17: positional reasoning</td>
<td valign="top" align="center">32.4 &#x000B1; 8</td>
<td valign="top" align="center">8.8 &#x000B1; 3.8</td>
<td valign="top" align="center">34.8 &#x000B1; 4.1</td>
<td valign="top" align="center">12.4 &#x000B1; 5.9</td>
<td valign="top" align="center">4.8 &#x000B1; 4.8</td>
<td valign="top" align="center">4.2 &#x000B1; 0.9</td>
<td valign="top" align="center"><bold>0.6</bold> <bold>&#x000B1;1.1</bold></td>
</tr>
<tr>
<td valign="top" align="left">18: size reasoning</td>
<td valign="top" align="center">4.2 &#x000B1; 1.8</td>
<td valign="top" align="center">1.3 &#x000B1; 0.9</td>
<td valign="top" align="center">5.0 &#x000B1; 1.4</td>
<td valign="top" align="center">1.6 &#x000B1; 1.1</td>
<td valign="top" align="center">0.4 &#x000B1; 0.4</td>
<td valign="top" align="center">0.4 &#x000B1; 0.2</td>
<td valign="top" align="center"><bold>0.0</bold> <bold>&#x000B1;0.0</bold></td>
</tr>
<tr>
<td valign="top" align="left">19: path finding</td>
<td valign="top" align="center">64.6 &#x000B1; 37.4</td>
<td valign="top" align="center">70.4 &#x000B1; 6.1</td>
<td valign="top" align="center">90.9 &#x000B1; 1.1</td>
<td valign="top" align="center">30.8 &#x000B1; 24.2</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center">0.4 &#x000B1; 0.8</td>
</tr>
<tr>
<td valign="top" align="left">20: agents motivation</td>
<td valign="top" align="center">0.0 &#x000B1; 0.1</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center">1.3 &#x000B1; 0.4</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center">0.1 &#x000B1; 0.1</td>
<td valign="top" align="center">0.0 &#x000B1; 0.0</td>
<td valign="top" align="center"><bold>0.0</bold> <bold>&#x000B1;0.0</bold></td>
</tr> <tr>
<td valign="top" align="left">Mean WER:</td>
<td valign="top" align="center">16.7 &#x000B1; 7.6</td>
<td valign="top" align="center">9.7 &#x000B1; 2.6</td>
<td valign="top" align="center">27.3 &#x000B1; 0.8</td>
<td valign="top" align="center">6.4 &#x000B1; 2.5</td>
<td valign="top" align="center">3.2 &#x000B1; 0.5</td>
<td valign="top" align="center">2.9 &#x000B1; 0.0</td>
<td valign="top" align="center"><bold>2.2</bold> <bold>&#x000B1;0.5</bold></td>
</tr>
<tr>
<td valign="top" align="left">Failed Tasks (&#x0003E;5%):</td>
<td valign="top" align="center">11.2 &#x000B1; 5.4</td>
<td valign="top" align="center">5.0 &#x000B1; 1.2</td>
<td valign="top" align="center">17.1 &#x000B1; 1.0</td>
<td valign="top" align="center">4.1 &#x000B1; 1.6</td>
<td valign="top" align="center">1.4 &#x000B1; 0.5</td>
<td valign="top" align="center">1.4 &#x000B1; 0.4</td>
<td valign="top" align="center"><bold>1.0</bold> <bold>&#x000B1;0.0</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values indicate the best performance in the comparison experiments.</p>
</table-wrap-foot>
</table-wrap>
<p>According to the experimental results, the MT-DNC model achieves a lower average error rate (2.2% mean WER) compared to other models, particularly the representative BrsDNC model, which demonstrates superior performance with a mean WER of 3.2% on the 20 bAbI sub-tasks under joint training. Specifically, for the 14<italic>th</italic>, 15<italic>th</italic>, and 18<italic>th</italic> sub-tasks, all other methods produce errors, while our method achieves an error rate of 0%. For the 16<italic>th</italic> and 17<italic>th</italic> sub-tasks, our method significantly reduces the error rate by 13.8% and 4.2%, respectively, compared to the BrsDNC model. Additionally, we counted the number of failed tasks (those with more than 5% errors) across the 20 sub-tasks, as shown in the last row of <xref ref-type="table" rid="T1">Table 1</xref>. Our method has only one failed task and outperforms other methods, significantly surpassing the DNC (with 11 failed tasks) and LSTM (with 17 failed tasks) models.</p>
<p><xref ref-type="fig" rid="F2">Figure 2</xref> illustrates the loss trends of different models during validation (<xref ref-type="fig" rid="F2">Figure 2A</xref>) and training (<xref ref-type="fig" rid="F2">Figure 2B</xref>) processes. As shown, the MT-DNC model demonstrates lower loss, higher performance, and faster convergence compared to the DNC and BrsDNC models. Furthermore, the variance of the learning curves in <xref ref-type="fig" rid="F2">Figures 2A</xref>, <xref ref-type="fig" rid="F2">B</xref> indicates that our method is more stable, with minimal fluctuations, while the BrsDNC model exhibits significant instability and fluctuating learning processes. Overall, our MT-DNC model improves convergence speed and performance while maintaining superior stability.</p>
<fig position="float" id="F2">
<label>Figure 2</label>
<caption><p>Validation loss <bold>(A)</bold> and training loss <bold>(B)</bold> of DNC, BrsDNC, MT-DNC-DI and MT-DNC. The horizontal axis represents the number of Epochs and the vertical axis represents the change of loss.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1635932-g0002.tif">
<alt-text>Two line charts compare model performance over epochs. Chart A shows valid loss decreasing across four models DNC, BrsDNC, MT-DNC, and MT-DNC-DI. Chart B depicts train loss decreasing similarly. Each model has distinct line colors.</alt-text>
</graphic>
</fig>
</sec>
<sec>
<title>4.4 Ablation study</title>
<p>To further analyze the validity of our proposed model, we conducted a series of ablation experiments. The main innovation of our model lies in the introduction of long-term memory and the memory transformation algorithm. In the MT-DNC model, the long-term memory module receives input from the working memory module through the memory transformation algorithm. To verify the effectiveness of the memory transformation mechanism, we compared the performance of MT-DNC and MT-DNC-DI. In the MT-DNC-DI model, the long-term memory module receives input directly from the controller layer (with different parameters from the working memory module). From <xref ref-type="table" rid="T1">Table 1</xref> and <xref ref-type="fig" rid="F2">Figure 2</xref>, we observe that MT-DNC achieves superior performance compared to MT-DNC-DI, both in terms of WER on each sub-task and in terms of average WER. Additionally, the MT-DNC-DI model performs better and exhibits lower loss compared to DNC, BrsDNC, and other models, indicating that the long-term memory itself contributes positively to model performance, while the memory transformation mechanism further enhances it.</p>
<p>We also analyzed the effect of storage space in long-term memory and working memory on the experimental results. <xref ref-type="fig" rid="F3">Figure 3</xref> illustrates the changes in mean WER during the learning process at different memory space sizes. We compared these results with the changes in mean WER of the BrsDNC model (black line in <xref ref-type="fig" rid="F3">Figure 3</xref>). The experimental results reveal that when the memory space is too small (e.g., 32 or 64), the performance of the model is negatively affected. Our model achieves comparable performance to the BrsDNC model under very small memory spaces (32 and 64), despite the BrsDNC model using a larger memory space of 128. However, our MT-DNC model significantly outperforms the BrsDNC model at memory space lengths of 128 and 256. Furthermore, we found that excessive memory space (e.g., 512) does not improve performance and instead leads to performance degradation. Overall, our model is robust and adaptable to different memory space lengths, but overly small or overly large memory spaces negatively impact performance compared to the most appropriate length.</p>
<fig position="float" id="F3">
<label>Figure 3</label>
<caption><p>Mean Word Error Rate of MT-DNC-32, MT-DNC-64, MT-DNC-128, MT-DNC-256, MT-DNC-512, BrsDNC. The horizontal coordinate represents the number of Epochs and the vertical coordinate represents the changing of Mean Word Error Rate.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1635932-g0003.tif">
<alt-text>Line graph showing the Mean Word Error Rate over 35 epochs for six models MT-DNC-32, MT-DNC-64, MT-DNC-128, MT-DNC-256, MT-DNC-512, and BrsDNC. Error rates decrease as epochs increase, with MT-DNC-512 achieving the lowest error rate. </alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec sec-type="conclusions" id="s5">
<title>5 Conclusion</title>
<p>In this paper, inspired by the memory transformation mechanism of the human brain, we propose the MT-DNC model, a coordinated framework with two memory modules: working memory and long-term memory. By establishing a connection between the working memory and the long-term memory, this model alleviates some of the challenges faced by DNCs. Specifically, as the amount of information in the memory region increases, the effectiveness of information retrieval and training efficiency improve, significantly impacting the model&#x00027;s convergence rate and final performance.</p>
<p>Nonetheless, several promising directions remain for future research. In particular, integrating the MT-DNC architecture with Transformer-based models is a key area of ongoing exploration. This hybrid approach aims to combine the structured, interpretable memory dynamics of MT-DNC with the powerful parallel processing capabilities of Transformers. By leveraging Transformer&#x00027;s inherent parallelism, the integrated model is expected to overcome the current limitations of sequential memory operations in DNC-based architectures, thereby improving computational efficiency and scalability.</p></sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found at: <ext-link ext-link-type="uri" xlink:href="https://github.com/Brain-Cog-Lab/MTDNC">https://github.com/Brain-Cog-Lab/MTDNC</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>YL: Conceptualization, Writing &#x02013; review &#x00026; editing, Methodology, Formal analysis, Writing &#x02013; original draft. YW: Funding acquisition, Writing &#x02013; original draft, Resources, Supervision, Validation, Writing &#x02013; review &#x00026; editing. HF: Validation, Conceptualization, Writing &#x02013; review &#x00026; editing, Writing &#x02013; original draft, Methodology. FZ: Supervision, Writing &#x02013; review &#x00026; editing, Methodology, Writing &#x02013; original draft, Conceptualization. YZ: Conceptualization, Funding acquisition, Project administration, Resources, Writing &#x02013; review &#x00026; editing, Supervision.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work was supported in part by the Beijing Major Science and Technology Project under Contract No. Z241100001324005.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p></sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn id="fn0001"><p><sup>1</sup><ext-link ext-link-type="uri" xlink:href="https://research.facebook.com/downloads/babi/">https://research.facebook.com/downloads/babi/</ext-link></p></fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Achiam</surname> <given-names>J.</given-names></name> <name><surname>Adler</surname> <given-names>S.</given-names></name> <name><surname>Agarwal</surname> <given-names>S.</given-names></name> <name><surname>Ahmad</surname> <given-names>L.</given-names></name> <name><surname>Akkaya</surname> <given-names>I.</given-names></name> <name><surname>Aleman</surname> <given-names>F. L.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Gpt-4 technical report</article-title>. <source>arXiv preprint arXiv:2303.08774</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2303.08774</pub-id></citation>
</ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Atkinson</surname> <given-names>R. C.</given-names></name> <name><surname>Shiffrin</surname> <given-names>R. M.</given-names></name></person-group> (<year>1968</year>). <article-title>Human memory: a proposed system and its control processes</article-title>. <source>Psychol. Learn. Motiv</source>. <volume>2</volume>, <fpage>89</fpage>&#x02013;<lpage>195</lpage>. <pub-id pub-id-type="doi">10.1016/S0079-7421(08)60422-3</pub-id></citation>
</ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ba</surname> <given-names>J.</given-names></name> <name><surname>Hinton</surname> <given-names>G. E.</given-names></name> <name><surname>Mnih</surname> <given-names>V.</given-names></name> <name><surname>Leibo</surname> <given-names>J. Z.</given-names></name> <name><surname>Ionescu</surname> <given-names>C.</given-names></name></person-group> (<year>2016</year>). <article-title>Using fast weights to attend to the recent past</article-title>. <source>Adv. Neural Inf. Process. Syst</source>. 29. <pub-id pub-id-type="doi">10.48550/arXiv.1610.06258</pub-id></citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ba</surname> <given-names>J. L.</given-names></name> <name><surname>Kiros</surname> <given-names>J. R.</given-names></name> <name><surname>Hinton</surname> <given-names>G. E.</given-names></name></person-group> (<year>2016</year>). <article-title>Layer normalization</article-title>. <source>arXiv preprint arXiv:1607.06450</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1607.06450</pub-id></citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Baddeley</surname> <given-names>A.</given-names></name></person-group> (<year>2007</year>). <source>Working Memory, Thought, and Action</source>. Oxford: Oxford University Press. <pub-id pub-id-type="doi">10.1093/acprof:oso/9780198528012.001.0001</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chan</surname> <given-names>A.</given-names></name> <name><surname>Ma</surname> <given-names>L.</given-names></name> <name><surname>Juefei-Xu</surname> <given-names>F.</given-names></name> <name><surname>Xie</surname> <given-names>X.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Ong</surname> <given-names>Y. S.</given-names></name></person-group> (<year>2018</year>). <article-title>Metamorphic relation based adversarial attacks on differentiable neural computer</article-title>. <source>arXiv preprint arXiv:1809.02444</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1809.02444</pub-id><pub-id pub-id-type="pmid">33886479</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Csord&#x000E1;s</surname> <given-names>R.</given-names></name> <name><surname>Schmidhuber</surname> <given-names>J.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Improving differentiable neural computers through memory masking, de-allocation, and link distribution sharpness control,&#x0201D;</article-title> in <source>International Conference on Learning Representations (ICLR)</source> (<publisher-loc>OpenReview.net</publisher-loc>).</citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Diamond</surname> <given-names>A.</given-names></name></person-group> (<year>2013</year>). <article-title>Executive functions</article-title>. <source>Annu. Rev. Psychol</source>. <volume>64</volume>:<fpage>135</fpage>. <pub-id pub-id-type="doi">10.1146/annurev-psych-113011-143750</pub-id><pub-id pub-id-type="pmid">23020641</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dubey</surname> <given-names>A.</given-names></name> <name><surname>Jauhri</surname> <given-names>A.</given-names></name> <name><surname>Pandey</surname> <given-names>A.</given-names></name> <name><surname>Kadian</surname> <given-names>A.</given-names></name> <name><surname>Al-Dahle</surname> <given-names>A.</given-names></name> <name><surname>Letman</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>The llama 3 herd of models</article-title>. <source>arXiv preprint arXiv:2407.21783</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2407.21783</pub-id></citation>
</ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Franke</surname> <given-names>J.</given-names></name> <name><surname>Niehues</surname> <given-names>J.</given-names></name> <name><surname>Waibel</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). <article-title>Robust and scalable differentiable neural computer for question answering</article-title>. <source>arXiv preprint arXiv:1807.02658</source>. <pub-id pub-id-type="doi">10.18653/v1/W18-2606</pub-id><pub-id pub-id-type="pmid">33886479</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Gal</surname> <given-names>Y.</given-names></name> <name><surname>Ghahramani</surname> <given-names>Z.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Dropout as a Bayesian approximation: representing model uncertainty in deep learning,&#x0201D;</article-title> in <source>Proceedings of the 33rd ICML</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>1050</fpage>&#x02013;<lpage>1059</lpage>.</citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Graves</surname> <given-names>A.</given-names></name> <name><surname>Wayne</surname> <given-names>G.</given-names></name> <name><surname>Danihelka</surname> <given-names>I.</given-names></name></person-group> (<year>2014</year>). <article-title>Neural turing machines</article-title>. <source>arXiv preprint arXiv:1410.5401</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1410.5401</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Graves</surname> <given-names>A.</given-names></name> <name><surname>Wayne</surname> <given-names>G.</given-names></name> <name><surname>Reynolds</surname> <given-names>M.</given-names></name> <name><surname>Harley</surname> <given-names>T.</given-names></name> <name><surname>Danihelka</surname> <given-names>I.</given-names></name> <name><surname>Grabska-Barwi&#x00144;ska</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Hybrid computing using a neural network with dynamic external memory</article-title>. <source>Nature</source> <volume>538</volume>, <fpage>471</fpage>&#x02013;<lpage>476</lpage>. <pub-id pub-id-type="doi">10.1038/nature20101</pub-id><pub-id pub-id-type="pmid">27732574</pub-id></citation></ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hassabis</surname> <given-names>D.</given-names></name> <name><surname>Kumaran</surname> <given-names>D.</given-names></name> <name><surname>Summerfield</surname> <given-names>C.</given-names></name> <name><surname>Botvinick</surname> <given-names>M.</given-names></name></person-group> (<year>2017</year>). <article-title>Neuroscience-inspired artificial intelligence</article-title>. <source>Neuron</source> <volume>95</volume>, <fpage>245</fpage>&#x02013;<lpage>258</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuron.2017.06.011</pub-id><pub-id pub-id-type="pmid">28728020</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Henaff</surname> <given-names>M.</given-names></name> <name><surname>Weston</surname> <given-names>J.</given-names></name> <name><surname>Szlam</surname> <given-names>A.</given-names></name> <name><surname>Bordes</surname> <given-names>A.</given-names></name> <name><surname>LeCun</surname> <given-names>Y.</given-names></name></person-group> (<year>2016</year>). <article-title>Tracking the world state with recurrent entity networks</article-title>. <source>arXiv preprint arXiv:1612.03969</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1612.03969</pub-id></citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hochreiter</surname> <given-names>S.</given-names></name> <name><surname>Schmidhuber</surname> <given-names>J.</given-names></name></person-group> (<year>1997</year>). <article-title>Long short-term memory</article-title>. <source>Neural Comput</source>. <volume>9</volume>, <fpage>1735</fpage>&#x02013;<lpage>1780</lpage>. <pub-id pub-id-type="doi">10.1162/neco.1997.9.8.1735</pub-id><pub-id pub-id-type="pmid">9377276</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hsin</surname> <given-names>C.</given-names></name></person-group> (<year>2016</year>). <source>Implementation and Optimization of Differentiable Neural Computers</source>. Technical Report.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ji</surname> <given-names>D.</given-names></name> <name><surname>Wilson</surname> <given-names>M. A.</given-names></name></person-group> (<year>2007</year>). <article-title>Coordinated memory replay in the visual cortex and hippocampus during sleep</article-title>. <source>Nat. Neurosci</source>. <volume>10</volume>, <fpage>100</fpage>&#x02013;<lpage>107</lpage>. <pub-id pub-id-type="doi">10.1038/nn1825</pub-id><pub-id pub-id-type="pmid">17173043</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kingma</surname> <given-names>D. P.</given-names></name> <name><surname>Ba</surname> <given-names>J. L.</given-names></name></person-group> (<year>2014</year>). <article-title>Adam: a method for stochastic optimization</article-title>. <source>arXiv preprint</source> arXiv:1412.6980.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kitamura</surname> <given-names>T.</given-names></name> <name><surname>Ogawa</surname> <given-names>S. K.</given-names></name> <name><surname>Roy</surname> <given-names>D. S.</given-names></name> <name><surname>Okuyama</surname> <given-names>T.</given-names></name> <name><surname>Morrissey</surname> <given-names>M. D.</given-names></name> <name><surname>Smith</surname> <given-names>L. M.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Engrams and circuits crucial for systems consolidation of a memory</article-title>. <source>Science</source> <volume>356</volume>, <fpage>73</fpage>&#x02013;<lpage>78</lpage>. <pub-id pub-id-type="doi">10.1126/science.aam6808</pub-id><pub-id pub-id-type="pmid">28386011</pub-id></citation></ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Klambauer</surname> <given-names>G.</given-names></name> <name><surname>Unterthiner</surname> <given-names>T.</given-names></name> <name><surname>Mayr</surname> <given-names>A.</given-names></name> <name><surname>Hochreiter</surname> <given-names>S.</given-names></name></person-group> (<year>2017</year>). <article-title>Self-normalizing neural networks</article-title>. <source>Adv. Neural Inf. Process. Syst</source>. 30.</citation>
</ref>
<ref id="B22">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kumar</surname> <given-names>A.</given-names></name> <name><surname>Irsoy</surname> <given-names>O.</given-names></name> <name><surname>Ondruska</surname> <given-names>P.</given-names></name> <name><surname>Iyyer</surname> <given-names>M.</given-names></name> <name><surname>Bradbury</surname> <given-names>J.</given-names></name> <name><surname>Gulrajani</surname> <given-names>I.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>&#x0201C;Ask me anything: dynamic memory networks for natural language processing,&#x0201D;</article-title> in <source>International Conference on Machine Learning</source> (<publisher-loc>PMLR</publisher-loc>), <fpage>1378</fpage>&#x02013;<lpage>1387</lpage>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lake</surname> <given-names>B. M.</given-names></name> <name><surname>Ullman</surname> <given-names>T. D.</given-names></name> <name><surname>Tenenbaum</surname> <given-names>J. B.</given-names></name> <name><surname>Gershman</surname> <given-names>S. J.</given-names></name></person-group> (<year>2017</year>). <article-title>Building machines that learn and think like people</article-title>. <source>Behav. Brain Sci</source>. <volume>40</volume>:<fpage>e253</fpage>. <pub-id pub-id-type="doi">10.1017/S0140525X16001837</pub-id><pub-id pub-id-type="pmid">27881212</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Le</surname> <given-names>H.</given-names></name> <name><surname>Tran</surname> <given-names>T.</given-names></name> <name><surname>Venkatesh</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>Learning to remember more with less memorization</article-title>. <source>arXiv preprint arXiv:1901.01347</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1901.01347</pub-id></citation>
</ref>
<ref id="B25">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Le</surname> <given-names>H.</given-names></name> <name><surname>Tran</surname> <given-names>T.</given-names></name> <name><surname>Venkatesh</surname> <given-names>S.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Self-attentive associative memory,&#x0201D;</article-title> in <source>International Conference on Machine Learning</source> (<publisher-loc>PMLR</publisher-loc>), <fpage>5682</fpage>&#x02013;<lpage>5691</lpage>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>LeCun</surname> <given-names>Y.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>Hinton</surname> <given-names>G.</given-names></name></person-group> (<year>2015</year>). <article-title>Deep learning</article-title>. <source>Nature</source> <volume>521</volume>, <fpage>436</fpage>&#x02013;<lpage>444</lpage>. <pub-id pub-id-type="doi">10.1038/nature14539</pub-id><pub-id pub-id-type="pmid">26017442</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>A. K.</given-names></name> <name><surname>Wilson</surname> <given-names>M. A.</given-names></name></person-group> (<year>2002</year>). <article-title>Memory of sequential experience in the hippocampus during slow wave sleep</article-title>. <source>Neuron</source> <volume>36</volume>, <fpage>1183</fpage>&#x02013;<lpage>1194</lpage>. <pub-id pub-id-type="doi">10.1016/S0896-6273(02)01096-6</pub-id><pub-id pub-id-type="pmid">12495631</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Marshall</surname> <given-names>L.</given-names></name> <name><surname>Born</surname> <given-names>J.</given-names></name></person-group> (<year>2007</year>). <article-title>The contribution of sleep to hippocampus-dependent memory consolidation</article-title>. <source>Trends Cogn. Sci</source>. <volume>11</volume>, <fpage>442</fpage>&#x02013;<lpage>450</lpage>. <pub-id pub-id-type="doi">10.1016/j.tics.2007.09.001</pub-id><pub-id pub-id-type="pmid">17905642</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rae</surname> <given-names>J.</given-names></name> <name><surname>Hunt</surname> <given-names>J. J.</given-names></name> <name><surname>Danihelka</surname> <given-names>I.</given-names></name> <name><surname>Harley</surname> <given-names>T.</given-names></name> <name><surname>Senior</surname> <given-names>A. W.</given-names></name> <name><surname>Wayne</surname> <given-names>G.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Scaling memory-augmented neural networks with sparse reads and writes</article-title>. <source>Adv. Neural Inf. Process. Syst</source>. 29.</citation>
</ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rasekh</surname> <given-names>M. S.</given-names></name> <name><surname>Safi-Esfahani</surname> <given-names>F.</given-names></name></person-group> (<year>2020</year>). <article-title>EDNC: evolving differentiable neural computers</article-title>. <source>Neurocomputing</source> <volume>412</volume>, <fpage>514</fpage>&#x02013;<lpage>542</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2020.06.018</pub-id></citation>
</ref>
<ref id="B31">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Santoro</surname> <given-names>A.</given-names></name> <name><surname>Bartunov</surname> <given-names>S.</given-names></name> <name><surname>Botvinick</surname> <given-names>M.</given-names></name> <name><surname>Wierstra</surname> <given-names>D.</given-names></name> <name><surname>Lillicrap</surname> <given-names>T.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Meta-learning with memory-augmented neural networks,&#x0201D;</article-title> in <source>International Conference on Machine Learning</source> (<publisher-loc>PMLR</publisher-loc>), <fpage>1842</fpage>&#x02013;<lpage>1850</lpage>.<pub-id pub-id-type="pmid">40674788</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Seo</surname> <given-names>M.</given-names></name> <name><surname>Min</surname> <given-names>S.</given-names></name> <name><surname>Farhadi</surname> <given-names>A.</given-names></name> <name><surname>Hajishirzi</surname> <given-names>H.</given-names></name></person-group> (<year>2016</year>). <article-title>Query-reduction networks for question answering</article-title>. <source>arXiv preprint arXiv:1606.04582</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1606.04582</pub-id></citation>
</ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Srivastava</surname> <given-names>N.</given-names></name> <name><surname>Hinton</surname> <given-names>G.</given-names></name> <name><surname>Krizhevsky</surname> <given-names>A.</given-names></name> <name><surname>Sutskever</surname> <given-names>I.</given-names></name> <name><surname>Salakhutdinov</surname> <given-names>R.</given-names></name></person-group> (<year>2014</year>). <article-title>Dropout: a simple way to prevent neural networks from overfitting</article-title>. <source>J. Mach. Learn. Res</source>. <volume>15</volume>, <fpage>1929</fpage>&#x02013;<lpage>1958</lpage>.<pub-id pub-id-type="pmid">33259321</pub-id></citation></ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tao</surname> <given-names>Q.</given-names></name> <name><surname>Xu</surname> <given-names>P.</given-names></name> <name><surname>Li</surname> <given-names>M.</given-names></name> <name><surname>Lu</surname> <given-names>W.</given-names></name></person-group> (<year>2021</year>). <article-title>Machine learning for perovskite materials design and discovery</article-title>. <source>NPJ Comput. Mater</source>. <volume>7</volume>, <fpage>1</fpage>&#x02013;<lpage>18</lpage>. <pub-id pub-id-type="doi">10.1038/s41524-021-00495-8</pub-id></citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Touvron</surname> <given-names>H.</given-names></name> <name><surname>Martin</surname> <given-names>L.</given-names></name> <name><surname>Stone</surname> <given-names>K.</given-names></name> <name><surname>Albert</surname> <given-names>P.</given-names></name> <name><surname>Almahairi</surname> <given-names>A.</given-names></name> <name><surname>Babaei</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Llama 2: Open foundation and fine-tuned chat models</article-title>. <source>arXiv preprint arXiv:2307.09288</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2307.09288</pub-id></citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Weston</surname> <given-names>J.</given-names></name> <name><surname>Bordes</surname> <given-names>A.</given-names></name> <name><surname>Chopra</surname> <given-names>S.</given-names></name> <name><surname>Rush</surname> <given-names>A. M.</given-names></name> <name><surname>Van Merri&#x000EB;nboer</surname> <given-names>B.</given-names></name> <name><surname>Joulin</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Towards AI-complete question answering: a set of prerequisite toy tasks</article-title>. <source>arXiv preprint arXiv:1502.05698</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1502.05698</pub-id></citation>
</ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Winocur</surname> <given-names>G.</given-names></name> <name><surname>Moscovitch</surname> <given-names>M.</given-names></name> <name><surname>Bontempi</surname> <given-names>B.</given-names></name></person-group> (<year>2010</year>). <article-title>Memory formation and long-term retention in humans and animals: convergence towards a transformation account of hippocampal-neocortical interactions</article-title>. <source>Neuropsychologia</source> <volume>48</volume>, <fpage>2339</fpage>&#x02013;<lpage>2356</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuropsychologia.2010.04.016</pub-id><pub-id pub-id-type="pmid">20430044</pub-id></citation></ref>
<ref id="B38">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Xiong</surname> <given-names>C.</given-names></name> <name><surname>Merity</surname> <given-names>S.</given-names></name> <name><surname>Socher</surname> <given-names>R.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Dynamic memory networks for visual and textual question answering,&#x0201D;</article-title> in <source>International Conference on Machine Learning</source> (<publisher-loc>PMLR</publisher-loc>), <fpage>2397</fpage>&#x02013;<lpage>2406</lpage>.</citation>
</ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zaremba</surname> <given-names>W.</given-names></name> <name><surname>Sutskever</surname> <given-names>I.</given-names></name></person-group> (<year>2015</year>). <article-title>Reinforcement learning neural turing machines-revised</article-title>. <source>arXiv preprint arXiv:1505.00521</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1505.00521</pub-id></citation>
</ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>Z.</given-names></name> <name><surname>Chen</surname> <given-names>W.</given-names></name> <name><surname>Wu</surname> <given-names>X.</given-names></name> <name><surname>Chen</surname> <given-names>P. C.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name></person-group> (<year>2017</year>). <article-title>LSTM network: a deep learning approach for short-term traffic forecast</article-title>. <source>IET Intell. Transp. Syst</source>. <volume>11</volume>, <fpage>68</fpage>&#x02013;<lpage>75</lpage>. <pub-id pub-id-type="doi">10.1049/iet-its.2016.0208</pub-id></citation>
</ref>
</ref-list>
<app-group>
<app id="A1">
<title>Appendix A</title>
<sec>
<title>Detailed Derivation of the Memory State Dimension ( S<sub><italic>t</italic></sub>)</title>
<p>The dimension <italic>S</italic><sub><italic>t</italic></sub> &#x0003D; (2<italic>R</italic>&#x0002B;6)<italic>W</italic>&#x0002B;6&#x0002B;4<italic>R</italic> is obtained by explicitly partitioning the output of the controller into signals required by the MT-DNC memory module operations. Below is the intuitive and step-by-step derivation aligned with the source code:</p>
<list list-type="order">
<list-item><p><bold>Working and Long-term Memory Writing Signals</bold></p></list-item>
</list>
<list list-type="bullet">
<list-item><p>Write keys for working and long-term memory: Each with dimension <italic>W</italic>, totaling 2<italic>W</italic>.</p></list-item>
<list-item><p>Write strengths (scalars) for both memories: 2 signals, each dimension 1, totaling 2.</p></list-item>
<list-item><p>Erase vectors for both memories: Each with dimension <italic>W</italic>, totaling 2<italic>W</italic>.</p></list-item>
<list-item><p>Write vectors for both memories: Each with dimension <italic>W</italic>, totaling 2<italic>W</italic>.</p></list-item>
<list-item><p>Allocation gates (scalars) for both memories: 2 signals, dimension 1 each, totaling 2.</p></list-item>
<list-item><p>Write gates (scalars) for both memories: 2 signals, dimension 1 each, totaling 2.</p></list-item>
</list>
<list list-type="simple">
<list-item><p><bold>Subtotal:</bold> 6<italic>W</italic>&#x0002B;6</p></list-item>
</list>
<list list-type="simple">
<list-item><p>2. <bold>Reading Signals (with multiple read heads</bold> <italic>R</italic><bold>)</bold></p></list-item>
</list>
<list list-type="bullet">
<list-item><p>Read keys for working and long-term memories: Each memory has <italic>R</italic> heads, each head dimension <italic>W</italic>, totaling 2<italic>RW</italic>.</p></list-item>
<list-item><p>Read strengths for working and long-term memories: Each memory has <italic>R</italic> read heads, each head a scalar, totaling 2<italic>R</italic>.</p></list-item>
<list-item><p>Free gates for working and long-term memories: Each memory has <italic>R</italic> read heads, each a scalar, totaling 2<italic>R</italic>.</p></list-item>
</list>
<list list-type="simple">
<list-item><p><bold>Subtotal:</bold> 2<italic>RW</italic>&#x0002B;4<italic>R</italic></p></list-item>
</list>
<list list-type="simple">
<list-item><p>3. <bold>Combine All Components</bold></p></list-item>
</list>
<disp-formula id="E12"><mml:math id="M86"><mml:mrow><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>6</mml:mn><mml:mi>W</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>6</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mi>R</mml:mi><mml:mi>W</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>4</mml:mn><mml:mi>R</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mi>R</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>6</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>W</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>6</mml:mn><mml:mo>&#x0002B;</mml:mo><mml:mn>4</mml:mn><mml:mi>R</mml:mi></mml:mrow></mml:math></disp-formula>
<p>This derivation matches precisely the dimensional partitioning provided in the implementation code as follows:</p>
<preformat>
&#x000A0;
&#x000A0;&#x000A0;write_keys:&#x000A0;[W]
&#x000A0;&#x000A0;write_keys_sec:&#x000A0;[W]&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x00023;&#x000A0;2W
&#x000A0;&#x000A0;write_strengths:&#x000A0;[1]
&#x000A0;&#x000A0;write_strengths_sec:&#x000A0;[1]&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x00023;&#x000A0;2
&#x000A0;&#x000A0;erase_vector:&#x000A0;[W]
&#x000A0;&#x000A0;erase_vector_sec:&#x000A0;[W]&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x00023;&#x000A0;2W
&#x000A0;&#x000A0;write_vector:&#x000A0;[W]
&#x000A0;&#x000A0;write_vector_sec:&#x000A0;[W]&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x00023;&#x000A0;2W
&#x000A0;&#x000A0;alloc_gates:&#x000A0;[1]
&#x000A0;&#x000A0;alloc_gates_sec:&#x000A0;[1]&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x00023;&#x000A0;2
&#x000A0;&#x000A0;write_gates:&#x000A0;[1]
&#x000A0;&#x000A0;write_gates_sec:&#x000A0;[1]&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x00023;&#x000A0;2
&#x000A0;&#x000A0;
&#x000A0;&#x000A0;&#x00023;&#x000A0;Total&#x000A0;so&#x000A0;far:&#x000A0;6W&#x000A0;&#x0002B;&#x000A0;6
&#x000A0;&#x000A0;
&#x000A0;&#x000A0;read_keys:&#x000A0;[R&#x000A0;x&#x000A0;W]
&#x000A0;&#x000A0;read_keys_sec:&#x000A0;[R&#x000A0;x&#x000A0;W]&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x00023;&#x000A0;2RW
&#x000A0;&#x000A0;read_strengths:&#x000A0;[R]
&#x000A0;&#x000A0;read_strengths_sec:&#x000A0;[R]&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x00023;&#x000A0;2R
&#x000A0;&#x000A0;free_gates:&#x000A0;[R]
&#x000A0;&#x000A0;free_gates_sec:&#x000A0;[R]&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x00023;&#x000A0;2R
&#x000A0;&#x000A0;
&#x000A0;&#x000A0;&#x00023;&#x000A0;Total&#x000A0;addition:&#x000A0;2RW&#x000A0;&#x0002B;&#x000A0;4R
&#x000A0;&#x000A0;
&#x000A0;&#x000A0;&#x00023;&#x000A0;Final&#x000A0;total:&#x000A0;(2R&#x000A0;&#x0002B;&#x000A0;6)W&#x000A0;&#x0002B;&#x000A0;6&#x000A0;&#x0002B;&#x000A0;4R
&#x000A0;
</preformat>
<p>Each component is explicitly represented and corresponds exactly to the signals used by the memory operation algorithms (writing, erasing, reading, gating), facilitating clear understanding and precise reproducibility of the MT-DNC architecture.</p>
</sec>
</app>
</app-group>
</back>
</article>