<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="brief-report">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neural Circuits</journal-id>
<journal-title>Frontiers in Neural Circuits</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neural Circuits</abbrev-journal-title>
<issn pub-type="epub">1662-5110</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fncir.2025.1618351</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Perspective</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Summary statistics of learning link changing neural representations to behavior</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Zavatone-Veth</surname> <given-names>Jacob A.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3047811/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Bordelon</surname> <given-names>Blake</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3175087/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Pehlevan</surname> <given-names>Cengiz</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="corresp" rid="c003"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/873694/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Center for Brain Science, Harvard University</institution>, <addr-line>Cambridge, MA</addr-line>, <country>United States</country></aff>
<aff id="aff2"><sup>2</sup><institution>Society of Fellows, Harvard University</institution>, <addr-line>Cambridge, MA</addr-line>, <country>United States</country></aff>
<aff id="aff3"><sup>3</sup><institution>John A. Paulson School of Engineering and Applied Sciences, Harvard University</institution>, <addr-line>Cambridge, MA</addr-line>, <country>United States</country></aff>
<aff id="aff4"><sup>4</sup><institution>Kempner Institute for the Study of Natural and Artificial Intelligence, Harvard University</institution>, <addr-line>Cambridge, MA</addr-line>, <country>United States</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Nicoletta Berardi, University of Florence, Italy</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Alexander van Meegen, Swiss Federal Institute of Technology Lausanne, Switzerland</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Jacob A. Zavatone-Veth <email>jzavatoneveth&#x00040;fas.harvard.edu</email></corresp>
<corresp id="c002">Blake Bordelon <email>blake_bordelon&#x00040;g.harvard.edu</email></corresp>
<corresp id="c003">Cengiz Pehlevan <email>cpehlevan&#x00040;seas.harvard.edu</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>29</day>
<month>08</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>19</volume>
<elocation-id>1618351</elocation-id>
<history>
<date date-type="received">
<day>25</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>11</day>
<month>08</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Zavatone-Veth, Bordelon and Pehlevan.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Zavatone-Veth, Bordelon and Pehlevan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>How can we make sense of large-scale recordings of neural activity across learning? Theories of neural network learning with their origins in statistical physics offer a potential answer: for a given task, there are often a small set of summary statistics that are sufficient to predict performance as the network learns. Here, we review recent advances in how summary statistics can be used to build theoretical understanding of neural network learning. We then argue for how this perspective can inform the analysis of neural data, enabling better understanding of learning in biological and artificial neural networks.</p></abstract>
<kwd-group>
<kwd>neural networks</kwd>
<kwd>learning</kwd>
<kwd>statistical physics</kwd>
<kwd>representation learning</kwd>
<kwd>summary statistics</kwd>
<kwd>representational similarity analysis</kwd>
</kwd-group>
<contract-num rid="cn002">DMS-2134157</contract-num>
<contract-sponsor id="cn001">National Institutes of Health<named-content content-type="fundref-id">https://doi.org/10.13039/100000002</named-content></contract-sponsor>
<contract-sponsor id="cn002">National Science Foundation<named-content content-type="fundref-id">https://doi.org/10.13039/100000001</named-content></contract-sponsor>
<contract-sponsor id="cn003">Defense Advanced Research Projects Agency<named-content content-type="fundref-id">https://doi.org/10.13039/100000185</named-content></contract-sponsor>
<counts>
<fig-count count="2"/>
<table-count count="0"/>
<equation-count count="12"/>
<ref-count count="75"/>
<page-count count="9"/>
<word-count count="7003"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Experience reshapes neural population activity, molding an animal&#x00027;s representations of the world as it learns to perform new tasks. Thanks to advances in experimental technologies, it is just now becoming possible to measure changes in the activity of large neural populations across the course of learning (<xref ref-type="bibr" rid="B44">Masset et al., 2022</xref>; <xref ref-type="bibr" rid="B21">Fink et al., 2025</xref>; <xref ref-type="bibr" rid="B40">Kriegeskorte and Wei, 2021</xref>; <xref ref-type="bibr" rid="B58">Steinmetz et al., 2021</xref>; <xref ref-type="bibr" rid="B75">Zhong et al., 2025</xref>; <xref ref-type="bibr" rid="B61">Sun et al., 2025</xref>; <xref ref-type="bibr" rid="B62">Vaidya et al., 2025</xref>). However, with this new capability comes the challenge of identifying which features of high-dimensional activity patterns are meaningful for understanding learning. While analyses of representations have begun how to elucidate how learning reshapes the structure of activity, it is not in general clear whether these measurements are sufficient to understand how representational changes relate to behavior (<xref ref-type="bibr" rid="B38">Krakauer et al., 2017</xref>; <xref ref-type="bibr" rid="B60">Sucholutsky et al., 2024</xref>; <xref ref-type="bibr" rid="B39">Kriegeskorte et al., 2008</xref>; <xref ref-type="bibr" rid="B40">Kriegeskorte and Wei, 2021</xref>).</p>
<p>In this Perspective, we propose that the principled identification of <bold>summary statistics of learning</bold> offers a possible path forward. This framework is grounded in theories of the statistical physics of learning in neural networks, which show that low-dimensional summary statistics are often sufficient to predict task performance over the course of learning (<xref ref-type="bibr" rid="B64">Watkin et al., 1993</xref>; <xref ref-type="bibr" rid="B19">Engel and van den Broeck, 2001</xref>; <xref ref-type="bibr" rid="B73">Zdeborov&#x000E1; and Krzakala, 2016</xref>). We argue that thinking systematically about summary statistics gives new insight into what existing approaches of quantifying neural representations reveal about learning, and allows identification of what additional measurements would be required to constrain models of plasticity. We emphasize that the goal of this Perspective is not to advocate for the use of a particular set of summary statistics, but rather to explain the general philosophy of this approach to understanding learning in high dimensions.</p></sec>
<sec id="s2">
<title>2 What is a summary statistic?</title>
<p>We posit that summary statistics of learning must satisfy two minimal desiderata:</p>
<list list-type="order">
<list-item><p><bold>They must be low-dimensional</bold>. That is, their dimension is low relative to the number of neurons in the network of interest. Indeed, most summary statistics we will encounter are determined by averages over the population of neurons.</p></list-item>
<list-item><p><bold>They must be sufficient to predict behavior across learning</bold>. From a theoretical standpoint, there should exist a closed set of equations describing the evolution of the summary statistics that predict the network&#x00027;s performance.</p></list-item>
</list>
<p>As we will illustrate with concrete examples in Section 3, summary statistics satisfying these two desiderata are often highly interpretable thanks to their clear relationship to the network architecture and learning task. However, the summary statistics relevant for predicting performance may not be sufficient to predict all statistical properties of population activity. We will elaborate on this issue, and the resulting limitations of descriptions based on summary statistics alone, in Section 4.</p>
<p>Our use of the term &#x0201C;summary statistics&#x0201D; follows work by <xref ref-type="bibr" rid="B4">Ben Arous et al. (2022</xref>, <xref ref-type="bibr" rid="B3">2023)</xref>. In the literature on the statistical physics of learning, the quantities that we refer to as summary statistics are often termed &#x0201C;order parameters&#x0201D; (<xref ref-type="bibr" rid="B45">M&#x000E9;zard et al., 1987</xref>; <xref ref-type="bibr" rid="B64">Watkin et al., 1993</xref>; <xref ref-type="bibr" rid="B19">Engel and van den Broeck, 2001</xref>; <xref ref-type="bibr" rid="B73">Zdeborov&#x000E1; and Krzakala, 2016</xref>). We prefer to use the former, more general term as it better captures the goal of these reduced descriptions in a neuroscientific context: we aim to summarize the features of neural activity relevant for learning.</p></sec>
<sec id="s3">
<title>3 Summary statistics in theories of neural network learning</title>
<p>We now review how summary statistics emerge naturally in theoretical analyses of neural network learning. Out of many theoretical results, we focus on two example settings: online learning from high-dimensional data in shallow networks, and batch learning in wide and deep networks (<xref ref-type="bibr" rid="B3">Ben Arous et al., 2023</xref>; <xref ref-type="bibr" rid="B23">Goldt et al., 2019</xref>; <xref ref-type="bibr" rid="B55">Saad and Solla, 1995</xref>; <xref ref-type="bibr" rid="B18">Cui et al., 2023</xref>; <xref ref-type="bibr" rid="B70">Zavatone-Veth and Pehlevan, 2021</xref>; <xref ref-type="bibr" rid="B11">Bordelon and Pehlevan, 2023b</xref>; <xref ref-type="bibr" rid="B72">Zavatone-Veth et al., 2022b</xref>; <xref ref-type="bibr" rid="B56">Saxe et al., 2013</xref>; <xref ref-type="bibr" rid="B7">Bordelon et al., 2025</xref>; <xref ref-type="bibr" rid="B1">Arnaboldi et al., 2023</xref>; <xref ref-type="bibr" rid="B63">van Meegen and Sompolinsky, 2025</xref>; <xref ref-type="bibr" rid="B64">Watkin et al., 1993</xref>; <xref ref-type="bibr" rid="B19">Engel and van den Broeck, 2001</xref>; <xref ref-type="bibr" rid="B73">Zdeborov&#x000E1; and Krzakala, 2016</xref>). These model problems illustrate how relevant summary statistics may be identified given a task, network architecture, and learning rule.</p>

<sec>
<title>3.1 Online learning in shallow neural networks with high dimensional data</title>
<p>Classical models of online gradient descent learning in high dimensions can be often be summarized with simple summary statistics (<xref ref-type="bibr" rid="B64">Watkin et al., 1993</xref>; <xref ref-type="bibr" rid="B19">Engel and van den Broeck, 2001</xref>; <xref ref-type="bibr" rid="B4">Ben Arous et al., 2022</xref>; <xref ref-type="bibr" rid="B1">Arnaboldi et al., 2023</xref>; <xref ref-type="bibr" rid="B23">Goldt et al., 2019</xref>, <xref ref-type="bibr" rid="B24">2020</xref>; <xref ref-type="bibr" rid="B6">Biehl and Schwarze, 1995</xref>; <xref ref-type="bibr" rid="B55">Saad and Solla, 1995</xref>). In this section, we discuss how the generalization performance of perceptrons and shallow (two-layer) neural networks trained on large quantities of high dimensional data can be summarized by simple weight alignment measures. Most simply, the perceptron model <inline-formula><mml:math id="M1"><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle><mml:mo>&#x000B7;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> seeks to learn a weight vector <bold>w</bold>&#x02208;&#x0211D;<sup><italic>D</italic></sup> which correctly classifies a finite set of randomly sampled training input-output pairs (<bold>x</bold><sub>&#x003BC;</sub>, <italic>y</italic><sub>&#x003BC;</sub>). If the inputs are random, <inline-formula><mml:math id="M2"><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0007E;</mml:mo><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>I</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, and the targets <italic>y</italic><sub>&#x003BC;</sub> &#x0003D; <italic>y</italic>(<bold>x</bold><sub>&#x003BC;</sub>) are generated by a <bold>teacher network</bold> <inline-formula><mml:math id="M3"><mml:mi>y</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, then the generalization performance (performance of the model on new <italic>unseen data</italic>, <inline-formula><mml:math id="M4"><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>y</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula>) is completely determined by the overlap of <bold>w</bold> with itself and with the target direction <bold>w</bold><sub>&#x022C6;</sub></p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>Q</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:mfrac><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle><mml:mo>&#x000B7;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle><mml:mtext>&#x000A0;</mml:mtext><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mi>R</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:mfrac><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msub><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>If the learning rate is scaled appropriately with the dimension <italic>D</italic>, the high-dimensional (large-<italic>D</italic>) limit of online stochastic gradient descent is given by a deterministic set of equations for <italic>Q</italic> and <italic>R</italic>:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>&#x003C4;</mml:mi></mml:mrow></mml:mfrac><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mi>Q</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>R</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>F</mml:mtext></mml:mstyle><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>Q</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mi>R</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where the continuous training &#x0201C;time&#x0201D; &#x003C4; is the ratio of the number of samples seen to the dimension and <bold>F</bold>:&#x0211D;<sup>2</sup> &#x02192; &#x0211D;<sup>2</sup> is a nonlinear function that depends on the learning rate, the loss function, and the link function &#x003C3;(&#x000B7;) (<xref ref-type="bibr" rid="B19">Engel and van den Broeck, 2001</xref>; <xref ref-type="bibr" rid="B4">Ben Arous et al., 2022</xref>; <xref ref-type="bibr" rid="B1">Arnaboldi et al., 2023</xref>; <xref ref-type="bibr" rid="B23">Goldt et al., 2019</xref>; <xref ref-type="bibr" rid="B55">Saad and Solla, 1995</xref>). Integrating this update equation allows one to predict the evolution of the generalization error as more training data are provided to the algorithm. Despite the infinite dimensionality of the original optimization problem, only two dimensions are necessary to capture the dynamics of generalization error.</p>
<p>The analysis of online perceptron learning can be extended to two layer neural networks with a small number of hidden neurons <italic>N</italic>,</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x000A0;</mml:mtext><mml:mi>&#x003D5;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mtext>&#x02003;</mml:mtext><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mtext>&#x000A0;</mml:mtext><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mi>i</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mi>N</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E4"><label>(4)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>y</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mtext>&#x02003;</mml:mtext><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msubsup><mml:mo>&#x000B7;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mtext>&#x000A0;</mml:mtext><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mi>k</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mi>K</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In this setting with isotropic random data, the relevant summary statistics are the readout weights <bold>a</bold>&#x02208;&#x0211D;<sup><italic>N</italic></sup>, along with <bold>overlap matrices</bold> <bold>Q</bold>&#x02208;&#x0211D;<sup><italic>N</italic>&#x000D7;<italic>N</italic></sup> and <bold>R</bold>&#x02208;&#x0211D;<sup><italic>N</italic>&#x000D7;<italic>K</italic></sup> with entries</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M9"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:mfrac><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x000A0;</mml:mtext><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:mfrac><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msubsup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>For this system, we can track the gradient descent dynamics for <bold>a</bold>, <bold>Q</bold>, and <bold>R</bold> through a generalization of <xref ref-type="disp-formula" rid="E2">Equation 2</xref> (<xref ref-type="bibr" rid="B23">Goldt et al., 2019</xref>; <xref ref-type="bibr" rid="B55">Saad and Solla, 1995</xref>; <xref ref-type="bibr" rid="B6">Biehl and Schwarze, 1995</xref>; <xref ref-type="bibr" rid="B24">Goldt et al., 2020</xref>). This reduces the dimensionality of the dynamics from the <italic>N</italic>&#x0002B;<italic>DN</italic> trainable parameters {<italic>a</italic><sub><italic>i</italic></sub>}, {<italic>w</italic><sub><italic>j</italic></sub>} to <italic>N</italic>&#x0002B;<italic>N</italic><sup>2</sup>&#x0002B;<italic>NK</italic> summary statistics, which is significant when <italic>D</italic>&#x0226B;<italic>N</italic>&#x0002B;<italic>K</italic>. This reduction enables the application of analyses that cannot scale to high dimensions, for instance control-theoretic methods to study optimal learning hyperparameters and curricula (<xref ref-type="bibr" rid="B49">Mori et al., 2025</xref>; <xref ref-type="bibr" rid="B46">Mignacco and Mori, 2025</xref>). Recent works have also begun to study approximations to these summary statistics when the network width <italic>N</italic> is also large, as further dimensionality reduction if possible when <bold>Q</bold> and <bold>R</bold> have stereotyped structures (<xref ref-type="bibr" rid="B48">Montanari and Urbani, 2025</xref>; <xref ref-type="bibr" rid="B1">Arnaboldi et al., 2023</xref>).</p>
<p>Under what conditions is this reduction possible? Fundamentally, the summary statistics <bold>a</bold>, <bold>Q</bold>, and <bold>R</bold> are sufficient to determine the network&#x00027;s performance so long as the preactivations <italic>h</italic><sub><italic>i</italic></sub> and <inline-formula><mml:math id="M10"><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msubsup></mml:math></inline-formula> are approximately Gaussian. Thus, one can relax the assumption that the inputs <bold>x</bold> are exactly Gaussian so long as a central limit theorem applies to <italic>h</italic><sub><italic>i</italic></sub> and <inline-formula><mml:math id="M11"><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msubsup></mml:math></inline-formula> (<xref ref-type="bibr" rid="B23">Goldt et al., 2019</xref>, <xref ref-type="bibr" rid="B24">2020</xref>). Moreover, one can allow for correlations between the different input dimensions so long as <italic>h</italic><sub><italic>i</italic></sub> and <inline-formula><mml:math id="M12"><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msubsup></mml:math></inline-formula> remain Gaussian. If <italic>E</italic>[<bold>xx</bold><sup>&#x022A4;</sup>] &#x0003D; <bold>&#x003A3;</bold>, with a modification of the definition of the overlaps to <inline-formula><mml:math id="M13"><mml:msub><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:mfrac><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>&#x003A3;</mml:mtext></mml:mstyle><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="M14"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:mfrac><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>&#x003A3;</mml:mtext></mml:mstyle><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msubsup></mml:math></inline-formula> a similar reduction applies (<xref ref-type="bibr" rid="B1">Arnaboldi et al., 2023</xref>). One can even consider extensions to plasticity rules other than stochastic gradient descent. For example, online node perturbation leads to a different effective dynamics for the same set of summary statistics (<xref ref-type="bibr" rid="B26">Hara et al., 2011</xref>, <xref ref-type="bibr" rid="B27">2013</xref>).</p>
<p>How could the overlaps <bold>Q</bold> and <bold>R</bold> be accessed from measurements of neural activity? And, in the absence of detailed knowledge of a teacher network, how could one identify the relevant overlaps? Under the simple structural assumptions of these models, one could estimate the overlaps from covariances of network activity across stimuli, i.e., with isotropic inputs one has <inline-formula><mml:math id="M15"><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022C6;</mml:mo></mml:mrow></mml:msubsup></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <italic>E</italic><sub><bold>x</bold></sub>[<italic>h</italic><sub><italic>i</italic></sub><italic>h</italic><sub><italic>j</italic></sub>] &#x0003D; <italic>Q</italic><sub><italic>ij</italic></sub>. Moreover, one can in some cases detect this underlying low-dimensional structure by examining the principal components of the learning trajectory (<xref ref-type="bibr" rid="B3">Ben Arous et al., 2023</xref>). However, more theoretical work is required in this vein.</p></sec>
<sec>
<title>3.2 Learning in wide and deep neural networks</title>
<p>Another strategy to reduce the complexity of multilayer deep neural networks is to analyze the dynamics of learning in terms of representational similarity matrices (kernels) for each hidden layer of the network. Consider, for example, a deep fully-connected network with input <bold>x</bold>&#x02208;&#x0211D;<sup><italic>D</italic></sup>,</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M16"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mi>f</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mi>&#x003B3;</mml:mi><mml:msqrt><mml:mi>N</mml:mi></mml:msqrt></mml:mrow></mml:mfrac><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:munderover><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mstyle><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mi>&#x003D5;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mi>h</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>L</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msubsup><mml:mi>h</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>&#x02113;</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:msqrt><mml:mi>N</mml:mi></mml:msqrt></mml:mrow></mml:mfrac><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:munderover><mml:mrow><mml:msubsup><mml:mi>W</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>&#x02113;</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup></mml:mrow></mml:mstyle><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mi>&#x003D5;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mi>h</mml:mi><mml:mi>j</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>&#x02113;</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:mi>&#x02113;</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mo>&#x0007B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:mi>L</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x0007D;</mml:mo><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:msubsup><mml:mi>h</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:msqrt><mml:mi>D</mml:mi></mml:msqrt></mml:mrow></mml:mfrac><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>D</mml:mi></mml:munderover><mml:mrow><mml:msubsup><mml:mi>W</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup></mml:mrow></mml:mstyle><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>t</italic> denotes training time. Instead of using online stochastic gradient descent to train the weights as we did in the preceding section, suppose we use gradient flow to minimize the average error on a fixed set of training examples. Moreover, instead of considering a regime where the hidden layer width <italic>N</italic> is small relative to the input dimension <italic>D</italic>, let us now consider very wide networks with <italic>N</italic>&#x0226B;<italic>D</italic> (<xref ref-type="fig" rid="F1">Figure 1a</xref>).</p>
<fig position="float" id="F1">
<label>Figure 1</label>
<caption><p>Representational similarity kernels in wide neural network models and in the brain. <bold>(a)</bold> Diagram of the infinite-width limit of a deep feedforward neural network. For a fixed input and output dimension, one considers a sequence of networks of increasing hidden layer widths. <bold>(b)</bold> Predicting the performance of width-2,500 fully-connected networks with three hidden layers and tanh activations over training using the dynamical mean-field theory described in Section 3. Networks are trained on a synthetic binary classification dataset of 10 examples, with 5 examples assigned each class at random. This leads to block structure in the final representations. Adapted from (<xref ref-type="bibr" rid="B11">Bordelon and Pehlevan 2023b</xref>). <bold>(c)</bold> The summary statistics in the dynamical mean field theory for the network in <bold>(b)</bold> are representational similarity kernels [<bold>&#x003A6;</bold><sup>(&#x02113;)</sup>; <italic>left</italic>] and gradient similarity kernels (<bold>G</bold><sup>&#x02113;</sup>; <italic>right</italic>) for each layer. The top row shows kernels estimated from gradient descent training, and the bottom row the theoretical predictions. All kernels are shown at the end of training (<italic>t</italic> &#x0003D; 100). Adapted from <xref ref-type="bibr" rid="B11">Bordelon and Pehlevan (2023b)</xref>. <bold>(d)</bold>. Comparing representational similarity kernels across models and brains. Here, similarity is measured using the Pearson correlation <italic>r</italic>, and the <italic>dissimilarity</italic> 1&#x02212;<italic>r</italic> is plotted as a heatmap. Kernels resulting from fMRI measurements of human inferior temporal (IT) cortex (<italic>left</italic>) and electrophysiological measurements of macaque monkey IT cortex (<italic>right</italic>) are compared with the kernel for features from a deep convolutional neural network after optimal re-weighting to match human IT (<italic>center</italic>). Adapted from Figure 10 of <xref ref-type="bibr" rid="B36">Khaligh-Razavi and Kriegeskorte (2014)</xref> with permission from N. Kriegeskorte under a CC-BY License.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncir-19-1618351-g0001.tif">
<alt-text>Illustration of neural network analysis. (a) Diagrams show increasing hidden layer width in networks over iterations t equals 1, 2, 3, up to infinity. (b) Graph of loss L(t) versus training time, comparing lazy and non-lazy neural networks with DMFT prediction. (c) Heatmaps depict experimental and theoretical representational similarity at different iterations t. (d) Heatmaps visualize representational dissimilarity in human IT, geometry-supervised deep convolutional network, and monkey IT, categorized by animate or inanimate and human or not human body/face features.</alt-text>
</graphic>
</fig>
<p>What are the relevant summary statistics in this case? Applying the chain rule to the dynamics of the network outputs, one finds the differential equation</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M17"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:mfrac><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>&#x02113;</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:msup><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x02113;</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msup><mml:mrow><mml:mi>&#x003A6;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x02113;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mfrac><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M18"><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:math></inline-formula> is the loss function and <inline-formula><mml:math id="M19"><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:math></inline-formula> denotes expectation over the training dataset (<xref ref-type="bibr" rid="B33">Jacot et al., 2018</xref>; <xref ref-type="bibr" rid="B41">Lee et al., 2019</xref>; <xref ref-type="bibr" rid="B11">Bordelon and Pehlevan, 2023b</xref>). Here,</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M20"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi>&#x003A6;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x02113;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mi>&#x003D5;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x02113;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>&#x003D5;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x02113;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>are <bold>representational similarity matrices</bold>, and</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M21"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x02113;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msubsup><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x02113;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E10"><label>(10)</label><mml:math id="M22"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x02113;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:msubsup><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x02113;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02261;</mml:mo><mml:mi>&#x003B3;</mml:mi><mml:msqrt><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msqrt><mml:mfrac><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x02113;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>are <bold>gradient similarity matrices</bold>, which respectively compare the hidden states <inline-formula><mml:math id="M23"><mml:mi>&#x003D5;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x02113;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> and the gradient signals <inline-formula><mml:math id="M24"><mml:msubsup><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x02113;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> at each hidden layer &#x02113; for each pair of data points (<bold>x</bold>, <bold>x</bold>&#x02032;) and each pair of training times (<italic>t, t</italic>&#x02032;). Thus, as &#x003A6;<sup>(&#x02113;)</sup> and <italic>G</italic><sup>(&#x02113;)</sup> determine the dynamics of <italic>f</italic>, these matrices are suitable summary statistics of learning if they are low-dimensional relative to the set of synaptic weights, and if we can write down a closed set of equations for their dynamics.</p>
<p>First, it is easy to see that the criterion of dimensionality reduction requires that the number of training examples <italic>P</italic> is much less than the network width <italic>N</italic>, as the number of similarity matrix elements and the number of synaptic weights are of order <italic>P</italic><sup>2</sup> and <italic>N</italic><sup>2</sup>, respectively. Second, it turns out that one can close the equations for &#x003A6;<sup>(&#x02113;)</sup> and <italic>G</italic><sup>(&#x02113;)</sup> provided that the width is large and that the synaptic weights start from an uninformed initial condition (i.e., Gaussian random matrices) (<xref ref-type="bibr" rid="B33">Jacot et al., 2018</xref>; <xref ref-type="bibr" rid="B41">Lee et al., 2019</xref>; <xref ref-type="bibr" rid="B68">Yang and Hu, 2021</xref>; <xref ref-type="bibr" rid="B11">Bordelon and Pehlevan, 2023b</xref>). Depending on how weights and learning rates are scaled, one can obtain different types of large-width (<italic>N</italic> &#x02192; &#x0221E;) limits (<xref ref-type="fig" rid="F1">Figure 1b</xref>). In the <italic>lazy/kernel</italic> limit where &#x003B3; is constant, these representational similarity matrices are static over the course of learning (<xref ref-type="bibr" rid="B33">Jacot et al., 2018</xref>; <xref ref-type="bibr" rid="B41">Lee et al., 2019</xref>). However, an alternative scaling (<inline-formula><mml:math id="M25"><mml:mi>&#x003B3;</mml:mi><mml:mo>&#x0221D;</mml:mo><mml:msqrt><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msqrt></mml:math></inline-formula>) can be adopted where these objects evolve in a task-dependent manner even as <italic>N</italic> &#x02192; &#x0221E; (<xref ref-type="fig" rid="F1">Figure 1c</xref>) (<xref ref-type="bibr" rid="B68">Yang and Hu, 2021</xref>; <xref ref-type="bibr" rid="B11">Bordelon and Pehlevan, 2023b</xref>).</p>
<p>While this provides a description of the training dynamics of a model under gradient flow, one can extend this description in terms of similarity matrices to other learning rules which use approximations of the backward pass variables <inline-formula><mml:math id="M26"><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x02113;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, which we called pseudo-gradients in <xref ref-type="bibr" rid="B10">Bordelon and Pehlevan (2023a)</xref>. Such rules include Hebbian learning, feedback alignment, and direct feedback alignment (<xref ref-type="bibr" rid="B30">Hebb, 2005</xref>; <xref ref-type="bibr" rid="B42">Lillicrap et al., 2016</xref>; <xref ref-type="bibr" rid="B51">N&#x000F8;kland, 2016</xref>). In this case, the relevant summary statistics to characterize the prediction dynamics of the network include the gradient-pseudogradient correlation, which measures the alignment between the gradients used by the learning rule and the gradients that one would have used with gradient flow,</p>
<disp-formula id="E11"><label>(11)</label><mml:math id="M27"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x02113;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msubsup><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x02113;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x02113;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>as <inline-formula><mml:math id="M28"><mml:msup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x02113;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:math></inline-formula> governs the evolution of the function output:</p>
<disp-formula id="E12"><label>(12)</label><mml:math id="M29"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:mfrac><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>&#x02113;</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:msup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x02113;</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msup><mml:mrow><mml:mi>&#x003A6;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x02113;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mfrac><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec></sec>
<sec id="s4">
<title>4 Implications for neural measurements</title>
<p>The two example settings detailed in Section 3 show how the relevant summary statistics of learning depend on network architecture and learning rule. Theoretical studies are just beginning to map out the full space of possible summary statistics for different network architectures (<xref ref-type="bibr" rid="B3">Ben Arous et al., 2023</xref>; <xref ref-type="bibr" rid="B23">Goldt et al., 2019</xref>; <xref ref-type="bibr" rid="B55">Saad and Solla, 1995</xref>; <xref ref-type="bibr" rid="B18">Cui et al., 2023</xref>; <xref ref-type="bibr" rid="B70">Zavatone-Veth and Pehlevan, 2021</xref>; <xref ref-type="bibr" rid="B11">Bordelon and Pehlevan, 2023b</xref>; <xref ref-type="bibr" rid="B72">Zavatone-Veth et al., 2022b</xref>; <xref ref-type="bibr" rid="B56">Saxe et al., 2013</xref>; <xref ref-type="bibr" rid="B7">Bordelon et al., 2025</xref>; <xref ref-type="bibr" rid="B1">Arnaboldi et al., 2023</xref>; <xref ref-type="bibr" rid="B63">van Meegen and Sompolinsky, 2025</xref>; <xref ref-type="bibr" rid="B19">Engel and van den Broeck, 2001</xref>; <xref ref-type="bibr" rid="B73">Zdeborov&#x000E1; and Krzakala, 2016</xref>). Though details of the relevant summary statistics vary depending on the scaling regime and task&#x02014;as illustrated by the examples above, where network width, training dataset size, and learning rule change the relevant statistics and their effective dynamics&#x02014;they share broad structural principles. In all cases, summary statistics are defined by (weighted) averages over sub-populations of neurons within the network of interest, e.g., correlations of activity with task-relevant variables, or autocorrelations of activity within a particular layer in a deep network. Thanks to these common structural features, these varied theories of summary statistics have common implications for the analysis and interpretation of neuroscience experiments.</p>
<sec>
<title>4.1 Benign sub-sampling</title>
<p>The summary statistics encountered in Section 3 are robust to sub-sampling thanks to their basic nature as averages over the population of neurons. These statistical theories in fact post a far stronger notion of benign sub-sampling: they result in neurons that are statistically exchangeable. This is highly advantageous from the perspective of long-term recordings of neural activity, as reliable measurement of summary statistics does not require one to track the exact same neurons over time. Instead, it suffices to measure a sufficiently large subpopulation on any given day. This obviates many of the challenges presented by tracking neurons over multiple recording sessions (<xref ref-type="bibr" rid="B44">Masset et al., 2022</xref>). Moreover, the variability and bias introduced by estimating summary statistics from a limited subset of relevant neurons can be characterized systematically (<xref ref-type="bibr" rid="B35">Kang et al., 2025</xref>; <xref ref-type="bibr" rid="B12">Bordelon and Pehlevan, 2024</xref>; <xref ref-type="bibr" rid="B69">Zavatone-Veth et al., 2022a</xref>). Taken together, these properties mean that summary statistics are relatively easy to estimate given limited neural measurements, provided that exchangability is not too strongly violated (<xref ref-type="bibr" rid="B22">Gao et al., 2017</xref>). We will return to this question in the Discussion, as a detailed analysis of the effects of non-identical neurons will be an important topic for future theoretical work. There are limits, however, to how far one can sub-sample. For instance, representational similarity kernels are more affected by small, coordinated changes in the tuning of many neurons than large changes in single-neuron tuning (<xref ref-type="fig" rid="F2">Figure 2</xref>) (<xref ref-type="bibr" rid="B40">Kriegeskorte and Wei, 2021</xref>). Determining the minimum number of neurons one must record in order to predict generalization dynamics across learning will be an important subject for future theoretical work (<xref ref-type="bibr" rid="B22">Gao et al., 2017</xref>; <xref ref-type="bibr" rid="B40">Kriegeskorte and Wei, 2021</xref>).</p>
<fig position="float" id="F2">
<label>Figure 2</label>
<caption><p>Invariance and universality in summary statistics. <bold>(a)</bold> Stable summary statistics despite drifting single-neuron responses. In <xref ref-type="bibr" rid="B53">Qin et al. (2023)</xref>&#x00027;s model of representational drift, single neurons are strongly tuned to a spatial variable, yet their tuning changes dramatically over time (<italic>left</italic>). Despite this drift, the similarity of the population representations of different spatial positions remains nearly constant (<italic>right</italic>). Adapted from Figure 5e of <xref ref-type="bibr" rid="B53">Qin et al. (2023)</xref>, of which C.P. is the corresponding author. <bold>(b)</bold> Universality of summary statistics in wide and deep networks with respect to the distribution of initial weights. Setting is as in <xref ref-type="fig" rid="F1">Figures 1b</xref>, <xref ref-type="fig" rid="F1">c</xref>, but also including a network for which the weights are initially drawn from {&#x02212;1, &#x0002B;1} with equal probability. Here, <italic>N</italic> &#x0003D; 2, 000, and a different realization of the random task is sampled relative to <xref ref-type="fig" rid="F1">Figures 1b</xref>, c, so the loss curves are not identical. <bold>(c)</bold> Cumulative distribution of weights at the start (<italic>initial</italic>) and end (<italic>final</italic>) of training for the networks shown in <bold>(b)</bold>. Note that the small change in the weight distributions for the Gaussian-initialized networks is not visible at this resolution, and that one expects the size of weight changes to scale with <inline-formula><mml:math id="M30"><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:msqrt><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msqrt></mml:math></inline-formula> (<xref ref-type="bibr" rid="B11">Bordelon and Pehlevan, 2023b</xref>). <bold>(d)</bold> Feature and gradient kernels at the end of training for the networks in <bold>(b)</bold>. No substantial differences are visible between networks initialized with different weight distributions.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncir-19-1618351-g0002.tif">
<alt-text>Graphical analysis of neural network training dynamics. Panel (a) shows neuron tuning curves and representational similarity matrices at two different times. Panel (b) depicts a loss curve over training time with different strategies: lazy and non-lazy, Gaussian and binary initialization, compared with DMFT prediction. Panel (c) illustrates cumulative frequency distributions of weight values over time. Panel (d) presents representational similarity matrices using different initialization strategies across three layers, highlighting maximum and minimum kernel values.</alt-text>
</graphic>
</fig></sec>
<sec>
<title>4.2 Invariances and representational drift</title>
<p>Though by our definition the summary statistics mentioned in Section 3 are sufficient to predict the network&#x00027;s performance, they are not sufficient statistics for all properties of the neural code. In particular, in part because they arise from theories in which neurons become exchangable, they have many invariances. These invariances mean that individual tuning curves can change substantially without altering the population-level computation (<xref ref-type="bibr" rid="B40">Kriegeskorte and Wei, 2021</xref>). For instance, the representational similarity kernels are invariant under rotation of the neural code at each layer, enabling complete reorganization of the single-neuron code without any effect on behavior. Similarly, overlaps with task-relevant directions are invariant to changes in the null space of those low-dimensional projections. These invariances mean that focusing on summary statistics of learning sets a particular aperture on what aspects of representations one can assay.</p>
<p>At the same time, the invariances of summary statistics have important consequences for functional robustness. In particular, they are closely related to theories of representational drift, the seemingly puzzling phenomenon of continuing changes in neural representations of task-relevant variables despite stable behavioral performance (<xref ref-type="bibr" rid="B54">Rule et al., 2019</xref>; <xref ref-type="bibr" rid="B44">Masset et al., 2022</xref>). Many models of drift explicitly propose that representational changes are structured in such a way that certain summary statistics are preserved (<xref ref-type="fig" rid="F2">Figure 2a</xref>) (<xref ref-type="bibr" rid="B44">Masset et al., 2022</xref>; <xref ref-type="bibr" rid="B52">Pashakhanloo and Koulakov, 2023</xref>; <xref ref-type="bibr" rid="B53">Qin et al., 2023</xref>). Identifying the invariances of the summary statistics sufficient to determine task performance can allow for a more systematic characterization of what forms of drift can be accommodated by a given network. Conversely, identifying the invariances of a representation once task performance stabilizes might suggest which summary statistics are relevant for the learning problem at hand.</p></sec>
<sec>
<title>4.3 Universality</title>
<p>An important lesson from the theory of high-dimensional statistics is that of <italic>universality</italic>: certain coarse-grained statistics are asymptotically insensitive to the details of the distribution. The most prominent example of statistical universality is the familiar central limit theorem: the distribution of the sample mean of independent random variables tends to a Gaussian as the number of samples becomes large. A broader class of universality principles arise in random matrix theory: the distribution of eigenvalues and eigenvectors of a random matrix often become insensitive to details of the distribution of the elements as the matrix becomes large. Most famously, the Mar&#x0010D;enko-Pastur theorem specifies that the singular values of a matrix with independent elements have a distribution that depends only on the mean and variance of the elements (<xref ref-type="bibr" rid="B43">Marchenko and Pastur, 1967</xref>). In the context of learning problems, universality manifests through insensitivity of the model performance to details of the distributions of parameters or of features (<xref ref-type="bibr" rid="B32">Hu and Lu, 2022</xref>; <xref ref-type="bibr" rid="B47">Misiakiewicz and Saeed, 2024</xref>).</p>
<p>From the perspective of summary statistics, statistical universality can allow simple theories to make informative macroscopic predictions even if they do not capture detailed properties of single neurons. For instance, the mean-field description of the learning dynamics of wide neural networks introduced in Section 3 are universal in that they depend on the initial distribution of hidden layer weights only through its mean and variance, even though the details of that distribution will affect the distribution of weights throughout training (<xref ref-type="fig" rid="F2">Figures 2b</xref>&#x02013;<xref ref-type="fig" rid="F2">d</xref>) (<xref ref-type="bibr" rid="B25">Golikov and Yang, 2022</xref>; <xref ref-type="bibr" rid="B67">Williams, 1996</xref>). Like the invariances to transformations of the neural population code mentioned before, this is nonetheless a double-edged sword: these universality properties mean that focusing on predicting performance commits one to coarse-graining away certain microscopic aspects of neural activity. Though these features are not required to predict macroscopic behavior, they may be important for understanding biological mechanisms.</p></sec></sec>
<sec sec-type="discussion" id="s5">
<title>5 Discussion</title>
<p>The core insight of the statistical mechanics of learning in neural networks is the existence of low-dimensional summary statistics sufficient to predict behavioral performance. We have reviewed how different summary statistics emerge depending on network architecture and task, how summary statistics might be estimated from experimental recordings, and what this perspective reveals about existing approaches to quantifying representational changes over learning. We now conclude by discussing complementary summary statistics of neural representations that arise from alternative desiderata, and future directions for theoretical inquiry.</p>
<p>A significant line of recent work in neuroscience aims to quantify neural representations and compare them across networks through analysis of representational similarity matrices &#x003A6;<sup>(&#x02113;)</sup>(<bold>x</bold>, <bold>x</bold>&#x02032;) (<xref ref-type="bibr" rid="B39">Kriegeskorte et al., 2008</xref>; <xref ref-type="bibr" rid="B60">Sucholutsky et al., 2024</xref>; <xref ref-type="bibr" rid="B66">Williams et al., 2021</xref>; <xref ref-type="bibr" rid="B65">Williams, 2024</xref>). Here, we see that these kernel matrices arise naturally as summary statistics of forward signal propagation in wide and deep neural networks (<xref ref-type="fig" rid="F1">Figures 1c</xref>, <xref ref-type="fig" rid="F1">d</xref>). At the same time, those results show that tracking only feature kernels is not in general sufficient to predict performance over the course of learning. One needs access also to coarse-grained information about the plasticity rule in the form of gradient kernels [either <italic>G</italic><sup>(&#x02113;)</sup> or <inline-formula><mml:math id="M31"><mml:msup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x02113;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:math></inline-formula>], and to information about the network outputs (for instance <inline-formula><mml:math id="M32"><mml:mi>&#x02202;</mml:mi><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow><mml:mo>/</mml:mo><mml:mi>&#x02202;</mml:mi><mml:mi>f</mml:mi></mml:math></inline-formula>). More theoretical work is required to determine how to reliably estimate these gradient kernels from data, thereby providing a means to gain coarse-grained information about the underlying plasticity rule.</p>
<p>The summary statistics discussed here explicitly depend on the architecture and nature of plasticity in the neural network of interest, as they seek to predict its performance over learning. A distinct set of summary statistics arises if one aims to study what features of a representation are relevant for an <italic>independently-trained</italic> decoder. In this line of work, one regards the representation as fixed, rather than considering end-to-end training of the full network as we considered here. If the decoder is a simple linear regressor that predicts a continuous variable, the relevant summary statistics of the representation are just its mean and covariance across stimuli (<xref ref-type="bibr" rid="B32">Hu and Lu, 2022</xref>; <xref ref-type="bibr" rid="B47">Misiakiewicz and Saeed, 2024</xref>). Given a particular task, the covariance can be further distilled into the rate of decay of its eigenvalues and of the projections of the task direction into its eigenvectors (<xref ref-type="bibr" rid="B29">Hastie et al., 2022</xref>; <xref ref-type="bibr" rid="B9">Bordelon and Pehlevan, 2022</xref>; <xref ref-type="bibr" rid="B13">Canatar et al., 2021</xref>, <xref ref-type="bibr" rid="B14">2024</xref>; <xref ref-type="bibr" rid="B2">Atanasov et al., 2024</xref>; <xref ref-type="bibr" rid="B65">Williams, 2024</xref>; <xref ref-type="bibr" rid="B28">Harvey et al., 2024</xref>; <xref ref-type="bibr" rid="B8">Bordelon et al., 2023</xref>). For categorically-structured stimuli, a substantial body of work has elucidated the summary statistics that emerge from assuming that one wants to divide the data according to a random dichotomy (<xref ref-type="bibr" rid="B16">Chung et al., 2018</xref>; <xref ref-type="bibr" rid="B17">Cohen et al., 2020</xref>; <xref ref-type="bibr" rid="B5">Bernardi et al., 2020</xref>; <xref ref-type="bibr" rid="B20">Farrell et al., 2022</xref>; <xref ref-type="bibr" rid="B19">Engel and van den Broeck, 2001</xref>; <xref ref-type="bibr" rid="B71">Zavatone-Veth and Pehlevan, 2022</xref>; <xref ref-type="bibr" rid="B57">Sorscher et al., 2022</xref>; <xref ref-type="bibr" rid="B28">Harvey et al., 2024</xref>).</p>
<p>The models reviewed here are composed of exchangeable neurons, which simplifies the relevant summary statistics and renders them particularly robust to sub-sampling. However, the brain has rich structure that can affect which summary statistics are sufficient to track learning and how those summary statistics may be measured. Biological neural networks are embedded in space, and their connectivity and selectivity is shaped by spatial structure (<xref ref-type="bibr" rid="B37">Khona et al., 2025</xref>; <xref ref-type="bibr" rid="B15">Chklovskii et al., 2002</xref>; <xref ref-type="bibr" rid="B59">Stiso and Bassett, 2018</xref>). Notably, many sensory areas are topographically organized: neurons with similar response properties are spatially proximal (<xref ref-type="bibr" rid="B34">Kandler et al., 2009</xref>; <xref ref-type="bibr" rid="B50">Murthy, 2011</xref>). Moreover, neurons can be classified into genetically-identifiable cell types (<xref ref-type="bibr" rid="B74">Zhang et al., 2023</xref>), which may play distinct functional roles during learning (<xref ref-type="bibr" rid="B31">Hirokawa et al., 2019</xref>; <xref ref-type="bibr" rid="B21">Fink et al., 2025</xref>). Future theoretical work must contend with these biological complexities in order to determine the relevant summary statistics of learning subject to these constraints.</p></sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>No experimental data were analyzed or generated in the preparation of this Perspective. Simulations of wide neural networks in <xref ref-type="fig" rid="F1">Figures 1b</xref>, <xref ref-type="fig" rid="F1">c</xref>, <xref ref-type="fig" rid="F2">2b</xref>&#x02013;<xref ref-type="fig" rid="F2">d</xref> following (<xref ref-type="bibr" rid="B11">Bordelon and Pehlevan 2023b</xref>) are based on code available under an MIT License at <ext-link ext-link-type="uri" xlink:href="https://github.com/Pehlevan-Group/dmft_wide_networks">https://github.com/Pehlevan-Group/dmft_wide_networks</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>JZ-V: Conceptualization, Funding acquisition, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. BB: Conceptualization, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. CP: Conceptualization, Funding acquisition, Writing &#x02013; review &#x00026; editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. JZ-V is supported by the Office of the Director of the National Institutes of Health under Award Number DP5OD037354. JZ-V is further supported by a Junior Fellowship from the Harvard Society of Fellows. BB is supported by a Google PhD Fellowship. CP is supported by NSF grant DMS-2134157, NSF CAREER Award IIS-2239780, DARPA grant DIAL-FP-038, a Sloan Research Fellowship, and The William F. Milton Fund from Harvard University. This work has been made possible in part by a gift from the Chan Zuckerberg Initiative Foundation to establish the Kempner Institute for the Study of Natural and Artificial Intelligence.</p>
</sec>
<ack>
<p>We are indebted to Nikolaus Kriegeskorte for sharing Figure 10 of (<xref ref-type="bibr" rid="B36">Khaligh-Razavi and Kriegeskorte 2014</xref>), from which our <xref ref-type="fig" rid="F1">Figure 1d</xref> is derived. We thank Paul Masset, Venkatesh Murthy, Farhad Pashakhanloo, and Ningjing Xia for helpful discussions and comments on previous versions of this manuscript.</p>
</ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p></sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Author disclaimer</title>
<p>The content is solely the responsibility of the authors and does not necessarily represent the official views of the National Institutes of Health.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Arnaboldi</surname> <given-names>L.</given-names></name> <name><surname>Stephan</surname> <given-names>L.</given-names></name> <name><surname>Krzakala</surname> <given-names>F.</given-names></name> <name><surname>Loureiro</surname> <given-names>B.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;From high-dimensional &#x00026;mean-field dynamics to dimensionless ODEs: a unifying approach to SGD in two-layers networks,&#x0201D;</article-title> in <source>Proceedings of Thirty Sixth Conference on Learning Theory, volume 195 of Proceedings of Machine Learning Research</source>, eds. G. Neu, and L. Rosasco (PMLR), <fpage>1199</fpage>&#x02013;<lpage>1227</lpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Atanasov</surname> <given-names>A.</given-names></name> <name><surname>Zavatone-Veth</surname> <given-names>J. A.</given-names></name> <name><surname>Pehlevan</surname> <given-names>C.</given-names></name></person-group> (<year>2024</year>). <article-title>Scaling and renormalization in high-dimensional regression</article-title>. <source>arXiv preprint arXiv:2405.00592</source>.</citation>
</ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ben Arous</surname> <given-names>G.</given-names></name> <name><surname>Gheissari</surname> <given-names>R.</given-names></name> <name><surname>Huang</surname> <given-names>J.</given-names></name> <name><surname>Jagannath</surname> <given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>High-dimensional SGD aligns with emerging outlier eigenspaces</article-title>. <source>arXiv preprint arXiv:2310.03010</source>.</citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ben Arous</surname> <given-names>G.</given-names></name> <name><surname>Gheissari</surname> <given-names>R.</given-names></name> <name><surname>Jagannath</surname> <given-names>A.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;High-dimensional limit theorems for SGD: effective dynamics and critical scaling,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source>, eds. S. Koyejo, S. Mohamed, A. Agarwal, D. Belgrave, K. Cho, and A. Oh (Curran Associates, Inc.), <fpage>25349</fpage>&#x02013;<lpage>25362</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bernardi</surname> <given-names>S.</given-names></name> <name><surname>Benna</surname> <given-names>M. K.</given-names></name> <name><surname>Rigotti</surname> <given-names>M.</given-names></name> <name><surname>Munuera</surname> <given-names>J.</given-names></name> <name><surname>Fusi</surname> <given-names>S.</given-names></name> <name><surname>Salzman</surname> <given-names>C. D.</given-names></name></person-group> (<year>2020</year>). <article-title>The geometry of abstraction in the hippocampus and prefrontal cortex</article-title>. <source>Cell</source> <volume>183</volume>, <fpage>954</fpage>&#x02013;<lpage>967</lpage>.e21. <pub-id pub-id-type="doi">10.1016/j.cell.2020.09.031</pub-id><pub-id pub-id-type="pmid">33058757</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Biehl</surname> <given-names>M.</given-names></name> <name><surname>Schwarze</surname> <given-names>H.</given-names></name></person-group> (<year>1995</year>). <article-title>Learning by on-line gradient descent</article-title>. <source>J. Phys. A Math. Gen</source>. <volume>28</volume>:<fpage>643</fpage>. <pub-id pub-id-type="doi">10.1088/0305-4470/28/3/018</pub-id></citation>
</ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bordelon</surname> <given-names>B.</given-names></name> <name><surname>Cotler</surname> <given-names>J.</given-names></name> <name><surname>Pehlevan</surname> <given-names>C.</given-names></name> <name><surname>Zavatone-Veth</surname> <given-names>J. A.</given-names></name></person-group> (<year>2025</year>). <article-title>Dynamically learning to integrate in recurrent neural networks</article-title>. <source>arXiv preprint arXiv:2503.18754</source>.</citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bordelon</surname> <given-names>B.</given-names></name> <name><surname>Masset</surname> <given-names>P.</given-names></name> <name><surname>Kuo</surname> <given-names>H.</given-names></name> <name><surname>Pehlevan</surname> <given-names>C.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Loss dynamics of temporal difference reinforcement learning,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source>, eds. A. Oh, T. Naumann, A. Globerson, K. Saenko, M. Hardt, and S. Levine (Curran Associates, Inc.), <fpage>14469</fpage>&#x02013;<lpage>14496</lpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bordelon</surname> <given-names>B.</given-names></name> <name><surname>Pehlevan</surname> <given-names>C.</given-names></name></person-group> (<year>2022</year>). <article-title>Population codes enable learning from few examples by shaping inductive bias</article-title>. <source>Elife</source> <volume>11</volume>:<fpage>e78606</fpage>. <pub-id pub-id-type="doi">10.7554/eLife.78606</pub-id><pub-id pub-id-type="pmid">36524716</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bordelon</surname> <given-names>B.</given-names></name> <name><surname>Pehlevan</surname> <given-names>C.</given-names></name></person-group> (<year>2023a</year>). <article-title>&#x0201C;The influence of learning rule on representation dynamics in wide neural networks,&#x0201D;</article-title> in <source>The Eleventh International Conference on Learning Representations</source>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bordelon</surname> <given-names>B.</given-names></name> <name><surname>Pehlevan</surname> <given-names>C.</given-names></name></person-group> (<year>2023b</year>). <article-title>Self-consistent dynamical field theory of kernel evolution in wide neural networks</article-title>. <source>J. Stat. Mech</source>. <volume>2023</volume>:<fpage>114009</fpage>. <pub-id pub-id-type="doi">10.1088/1742-5468/ad01b0</pub-id></citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bordelon</surname> <given-names>B.</given-names></name> <name><surname>Pehlevan</surname> <given-names>C.</given-names></name></person-group> (<year>2024</year>). <article-title>Dynamics of finite width kernel and prediction fluctuations in mean field neural networks</article-title>. <source>J. Statist. Mech</source>. <volume>2024</volume>:<fpage>104021</fpage>. <pub-id pub-id-type="doi">10.1088/1742-5468/ad642b</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Canatar</surname> <given-names>A.</given-names></name> <name><surname>Bordelon</surname> <given-names>B.</given-names></name> <name><surname>Pehlevan</surname> <given-names>C.</given-names></name></person-group> (<year>2021</year>). <article-title>Spectral bias and task-model alignment explain generalization in kernel regression and infinitely wide neural networks</article-title>. <source>Nat. Commun</source>. <volume>12</volume>:<fpage>2914</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-021-23103-1</pub-id><pub-id pub-id-type="pmid">34006842</pub-id></citation></ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Canatar</surname> <given-names>A.</given-names></name> <name><surname>Feather</surname> <given-names>J.</given-names></name> <name><surname>Wakhloo</surname> <given-names>A.</given-names></name> <name><surname>Chung</surname> <given-names>S.</given-names></name></person-group> (<year>2024</year>). <article-title>&#x0201C;A spectral theory of neural prediction and alignment,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source>, 36.</citation>
</ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chklovskii</surname> <given-names>D. B.</given-names></name> <name><surname>Schikorski</surname> <given-names>T.</given-names></name> <name><surname>Stevens</surname> <given-names>C. F.</given-names></name></person-group> (<year>2002</year>). <article-title>Wiring optimization in cortical circuits</article-title>. <source>Neuron</source> <volume>34</volume>, <fpage>341</fpage>&#x02013;<lpage>347</lpage>. <pub-id pub-id-type="doi">10.1016/S0896-6273(02)00679-7</pub-id></citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chung</surname> <given-names>S.</given-names></name> <name><surname>Lee</surname> <given-names>D. D.</given-names></name> <name><surname>Sompolinsky</surname> <given-names>H.</given-names></name></person-group> (<year>2018</year>). <article-title>Classification and geometry of general perceptual manifolds</article-title>. <source>Phys. Rev. X</source> <volume>8</volume>:<fpage>031003</fpage>. <pub-id pub-id-type="doi">10.1103/PhysRevX.8.031003</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cohen</surname> <given-names>U.</given-names></name> <name><surname>Chung</surname> <given-names>S.</given-names></name> <name><surname>Lee</surname> <given-names>D. D.</given-names></name> <name><surname>Sompolinsky</surname> <given-names>H.</given-names></name></person-group> (<year>2020</year>). <article-title>Separability and geometry of object manifolds in deep neural networks</article-title>. <source>Nat. Commun</source>. <volume>11</volume>:<fpage>746</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-020-14578-5</pub-id><pub-id pub-id-type="pmid">32029727</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cui</surname> <given-names>H.</given-names></name> <name><surname>Krzakala</surname> <given-names>F.</given-names></name> <name><surname>Zdeborova</surname> <given-names>L.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Bayes-optimal learning of deep random networks of extensive-width,&#x0201D;</article-title> in <source>Proceedings of the 40th International Conference on Machine Learning, volume 202 of Proceedings of Machine Learning Research</source>, eds. A. Krause, E. Brunskill, K. Cho, B. Engelhardt, S. Sabato, and J. Scarlett (PMLR), <fpage>6468</fpage>&#x02013;<lpage>6521</lpage>.</citation>
</ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Engel</surname> <given-names>A.</given-names></name> <name><surname>van den Broeck</surname> <given-names>C.</given-names></name></person-group> (<year>2001</year>). <source>Statistical Mechanics of Learning</source>. Cambridge: Cambridge University Press. <pub-id pub-id-type="doi">10.1017/CBO9781139164542</pub-id></citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Farrell</surname> <given-names>M.</given-names></name> <name><surname>Bordelon</surname> <given-names>B.</given-names></name> <name><surname>Trivedi</surname> <given-names>S.</given-names></name> <name><surname>Pehlevan</surname> <given-names>C.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Capacity of group-invariant linear readouts from equivariant representations: How many objects can be linearly classified under all possible views?,&#x0201D;</article-title> in <source>International Conference on Learning Representations</source>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fink</surname> <given-names>A. J.</given-names></name> <name><surname>Muscinelli</surname> <given-names>S. P.</given-names></name> <name><surname>Wang</surname> <given-names>S.</given-names></name> <name><surname>Hogan</surname> <given-names>M. I.</given-names></name> <name><surname>English</surname> <given-names>D. F.</given-names></name> <name><surname>Axel</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>Experience-dependent reorganization of inhibitory neuron synaptic connectivity</article-title>. <source>bioRxiv.16.633450</source>. <pub-id pub-id-type="doi">10.1101/2025.01.16.633450</pub-id><pub-id pub-id-type="pmid">39868262</pub-id></citation></ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>P.</given-names></name> <name><surname>Trautmann</surname> <given-names>E.</given-names></name> <name><surname>Yu</surname> <given-names>B.</given-names></name> <name><surname>Santhanam</surname> <given-names>G.</given-names></name> <name><surname>Ryu</surname> <given-names>S.</given-names></name> <name><surname>Shenoy</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>A theory of multineuronal dimensionality, dynamics and measurement</article-title>. <source>bioRxiv, 214262</source>. <pub-id pub-id-type="doi">10.1101/214262</pub-id></citation>
</ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Goldt</surname> <given-names>S.</given-names></name> <name><surname>Advani</surname> <given-names>M.</given-names></name> <name><surname>Saxe</surname> <given-names>A. M.</given-names></name> <name><surname>Krzakala</surname> <given-names>F.</given-names></name> <name><surname>Zdeborov&#x000E1;</surname> <given-names>L.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Dynamics of stochastic gradient descent for two-layer neural networks in the teacher-student setup,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source>, 32. <pub-id pub-id-type="doi">10.1088/1742-5468/abc61e</pub-id><pub-id pub-id-type="pmid">34262607</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Goldt</surname> <given-names>S.</given-names></name> <name><surname>M&#x000E9;zard</surname> <given-names>M.</given-names></name> <name><surname>Krzakala</surname> <given-names>F.</given-names></name> <name><surname>Zdeborov&#x000E1;</surname> <given-names>L.</given-names></name></person-group> (<year>2020</year>). <article-title>Modeling the influence of data structure on learning in neural networks: the hidden manifold model</article-title>. <source>Phys. Rev. X</source> <volume>10</volume>:<fpage>041044</fpage>. <pub-id pub-id-type="doi">10.1103/PhysRevX.10.041044</pub-id></citation>
</ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Golikov</surname> <given-names>E.</given-names></name> <name><surname>Yang</surname> <given-names>G.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Non-gaussian tensor programs,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source>, eds. S. Koyejo, S. Mohamed, A. Agarwal, D. Belgrave, K. Cho, and A. Oh (Curran Associates, Inc.), <fpage>21521</fpage>&#x02013;<lpage>21533</lpage>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hara</surname> <given-names>K.</given-names></name> <name><surname>Katahira</surname> <given-names>K.</given-names></name> <name><surname>Okanoya</surname> <given-names>K.</given-names></name> <name><surname>Okada</surname> <given-names>M.</given-names></name></person-group> (<year>2011</year>). <article-title>Statistical mechanics of on-line node-perturbation learning</article-title>. <source>IPSJ Online Trans</source>. <volume>4</volume>, <fpage>23</fpage>&#x02013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.2197/ipsjtrans.4.23</pub-id></citation>
</ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hara</surname> <given-names>K.</given-names></name> <name><surname>Katahira</surname> <given-names>K.</given-names></name> <name><surname>Okanoya</surname> <given-names>K.</given-names></name> <name><surname>Okada</surname> <given-names>M.</given-names></name></person-group> (<year>2013</year>). <article-title>Statistical mechanics of node-perturbation learning for nonlinear perceptron</article-title>. <source>J. Phys. Soc. Japan</source> <volume>82</volume>:<fpage>054001</fpage>. <pub-id pub-id-type="doi">10.7566/JPSJ.82.054001</pub-id></citation>
</ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Harvey</surname> <given-names>S. E.</given-names></name> <name><surname>Lipshutz</surname> <given-names>D.</given-names></name> <name><surname>Williams</surname> <given-names>A. H.</given-names></name></person-group> (<year>2024</year>). <article-title>&#x0201C;What representational similarity measures imply about decodable information,&#x0201D;</article-title> in <source>UniReps: 2nd Edition of the Workshop on Unifying Representations in Neural Models</source>.</citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hastie</surname> <given-names>T.</given-names></name> <name><surname>Montanari</surname> <given-names>A.</given-names></name> <name><surname>Rosset</surname> <given-names>S.</given-names></name> <name><surname>Tibshirani</surname> <given-names>R. J.</given-names></name></person-group> (<year>2022</year>). <article-title>Surprises in high-dimensional ridgeless least squares interpolation</article-title>. <source>Ann. Stat</source>. <volume>50</volume>, <fpage>949</fpage>&#x02013;<lpage>986</lpage>. <pub-id pub-id-type="doi">10.1214/21-AOS2133</pub-id></citation>
</ref>
<ref id="B30">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Hebb</surname> <given-names>D. O.</given-names></name></person-group> (<year>2005</year>). <source>The Organization of Behavior: A Neuropsychological Theory</source>. <publisher-loc>London</publisher-loc>: <publisher-name>Psychology press</publisher-name>.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hirokawa</surname> <given-names>J.</given-names></name> <name><surname>Vaughan</surname> <given-names>A.</given-names></name> <name><surname>Masset</surname> <given-names>P.</given-names></name> <name><surname>Ott</surname> <given-names>T.</given-names></name> <name><surname>Kepecs</surname> <given-names>A.</given-names></name></person-group> (<year>2019</year>). <article-title>Frontal cortex neuron types categorically encode single decision variables</article-title>. <source>Nature</source> <volume>576</volume>, <fpage>446</fpage>&#x02013;<lpage>451</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-019-1816-9</pub-id><pub-id pub-id-type="pmid">31801999</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>H.</given-names></name> <name><surname>Lu</surname> <given-names>Y. M.</given-names></name></person-group> (<year>2022</year>). <article-title>Universality laws for high-dimensional learning with random features</article-title>. <source>IEEE Trans. Inf. Theory</source> <volume>69</volume>, <fpage>1932</fpage>&#x02013;<lpage>1964</lpage>. <pub-id pub-id-type="doi">10.1109/TIT.2022.3217698</pub-id></citation>
</ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jacot</surname> <given-names>A.</given-names></name> <name><surname>Gabriel</surname> <given-names>F.</given-names></name> <name><surname>Hongler</surname> <given-names>C.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Neural tangent kernel: convergence and generalization in neural networks,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source>, 31.</citation>
</ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kandler</surname> <given-names>K.</given-names></name> <name><surname>Clause</surname> <given-names>A.</given-names></name> <name><surname>Noh</surname> <given-names>J.</given-names></name></person-group> (<year>2009</year>). <article-title>Tonotopic reorganization of developing auditory brainstem circuits</article-title>. <source>Nat. Neurosci</source>. <volume>12</volume>, <fpage>711</fpage>&#x02013;<lpage>717</lpage>. <pub-id pub-id-type="doi">10.1038/nn.2332</pub-id><pub-id pub-id-type="pmid">19471270</pub-id></citation></ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kang</surname> <given-names>H.</given-names></name> <name><surname>Canatar</surname> <given-names>A.</given-names></name> <name><surname>Chung</surname> <given-names>S.</given-names></name></person-group> (<year>2025</year>). <article-title>Spectral analysis of representational similarity with limited neurons</article-title>. <source>arXiv preprint arXiv:2502.19648</source>.</citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Khaligh-Razavi</surname> <given-names>S.-M.</given-names></name> <name><surname>Kriegeskorte</surname> <given-names>N.</given-names></name></person-group> (<year>2014</year>). <article-title>Deep supervised, but not unsupervised, models may explain it cortical representation</article-title>. <source>PLoS Comput. Biol</source>. <volume>10</volume>, <fpage>1</fpage>&#x02013;<lpage>29</lpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1003915</pub-id><pub-id pub-id-type="pmid">25375136</pub-id></citation></ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Khona</surname> <given-names>M.</given-names></name> <name><surname>Chandra</surname> <given-names>S.</given-names></name> <name><surname>Fiete</surname> <given-names>I.</given-names></name></person-group> (<year>2025</year>). <article-title>Global modules robustly emerge from local interactions and smooth gradients</article-title>. <source>Nature</source> <volume>640</volume>, <fpage>155</fpage>&#x02013;<lpage>164</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-024-08541-3</pub-id><pub-id pub-id-type="pmid">39972140</pub-id></citation></ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krakauer</surname> <given-names>J. W.</given-names></name> <name><surname>Ghazanfar</surname> <given-names>A. A.</given-names></name> <name><surname>Gomez-Marin</surname> <given-names>A.</given-names></name> <name><surname>MacIver</surname> <given-names>M. A.</given-names></name> <name><surname>Poeppel</surname> <given-names>D.</given-names></name></person-group> (<year>2017</year>). <article-title>Neuroscience needs behavior: correcting a reductionist bias</article-title>. <source>Neuron</source> <volume>93</volume>, <fpage>480</fpage>&#x02013;<lpage>490</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuron.2016.12.041</pub-id><pub-id pub-id-type="pmid">28182904</pub-id></citation></ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kriegeskorte</surname> <given-names>N.</given-names></name> <name><surname>Mur</surname> <given-names>M.</given-names></name> <name><surname>Bandettini</surname> <given-names>P. A.</given-names></name></person-group> (<year>2008</year>). <article-title>Representational similarity analysis - connecting the branches of systems neuroscience</article-title>. <source>Front. Syst. Neurosci</source>. <volume>2</volume>:<fpage>249</fpage>. <pub-id pub-id-type="doi">10.3389/neuro.06.004.2008</pub-id><pub-id pub-id-type="pmid">19104670</pub-id></citation></ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kriegeskorte</surname> <given-names>N.</given-names></name> <name><surname>Wei</surname> <given-names>X.-X.</given-names></name></person-group> (<year>2021</year>). <article-title>Neural tuning and representational geometry</article-title>. <source>Nat. Rev. Neurosci</source>. <volume>22</volume>, <fpage>703</fpage>&#x02013;<lpage>718</lpage>. <pub-id pub-id-type="doi">10.1038/s41583-021-00502-3</pub-id><pub-id pub-id-type="pmid">34522043</pub-id></citation></ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>J.</given-names></name> <name><surname>Xiao</surname> <given-names>L.</given-names></name> <name><surname>Schoenholz</surname> <given-names>S.</given-names></name> <name><surname>Bahri</surname> <given-names>Y.</given-names></name> <name><surname>Novak</surname> <given-names>R.</given-names></name> <name><surname>Sohl-Dickstein</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>&#x0201C;Wide neural networks of any depth evolve as linear models under gradient descent,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source>, eds. H. Wallach, H. Larochelle, A. Beygelzimer, F. d&#x00027; Alch&#x000E9;-Buc, E. Fox, and R. Garnett (Curran Associates, Inc.). <pub-id pub-id-type="doi">10.1088/1742-5468/abc62b</pub-id></citation>
</ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lillicrap</surname> <given-names>T. P.</given-names></name> <name><surname>Cownden</surname> <given-names>D.</given-names></name> <name><surname>Tweed</surname> <given-names>D. B.</given-names></name> <name><surname>Akerman</surname> <given-names>C. J.</given-names></name></person-group> (<year>2016</year>). <article-title>Random synaptic feedback weights support error backpropagation for deep learning</article-title>. <source>Nat. Commun</source>. <volume>7</volume>:<fpage>13276</fpage>. <pub-id pub-id-type="doi">10.1038/ncomms13276</pub-id><pub-id pub-id-type="pmid">27824044</pub-id></citation></ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Marchenko</surname> <given-names>V. A.</given-names></name> <name><surname>Pastur</surname> <given-names>L. A.</given-names></name></person-group> (<year>1967</year>). <article-title>Distribution of eigenvalues for some sets of random matrices</article-title>. <source>Matematicheskii Sbornik</source> <volume>114</volume>, <fpage>507</fpage>&#x02013;<lpage>536</lpage>.</citation>
</ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Masset</surname> <given-names>P.</given-names></name> <name><surname>Qin</surname> <given-names>S.</given-names></name> <name><surname>Zavatone-Veth</surname> <given-names>J. A.</given-names></name></person-group> (<year>2022</year>). <article-title>Drifting neuronal representations: bug or feature?</article-title> <source>Biol. Cyber</source>. <volume>116</volume>, <fpage>253</fpage>&#x02013;<lpage>266</lpage>. <pub-id pub-id-type="doi">10.1007/s00422-021-00916-3</pub-id><pub-id pub-id-type="pmid">34993613</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>M&#x000E9;zard</surname> <given-names>M.</given-names></name> <name><surname>Parisi</surname> <given-names>G.</given-names></name> <name><surname>Virasoro</surname> <given-names>M. A.</given-names></name></person-group> (<year>1987</year>). <source>Spin Glass Theory and Beyond: An Introduction to the Replica Method and Its Applications</source>. World Scientific Publishing Company. <pub-id pub-id-type="doi">10.1142/0271</pub-id></citation>
</ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mignacco</surname> <given-names>F.</given-names></name> <name><surname>Mori</surname> <given-names>F.</given-names></name></person-group> (<year>2025</year>). <article-title>A statistical physics framework for optimal learning</article-title>. <source>arXiv preprint arXiv:2507.07907</source>.</citation>
</ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Misiakiewicz</surname> <given-names>T.</given-names></name> <name><surname>Saeed</surname> <given-names>B.</given-names></name></person-group> (<year>2024</year>). <article-title>A non-asymptotic theory of kernel ridge regression: deterministic equivalents, test error, and GCV estimator</article-title>. <source>arXiv preprint arXiv:2403.08938</source>.</citation>
</ref>
<ref id="B48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Montanari</surname> <given-names>A.</given-names></name> <name><surname>Urbani</surname> <given-names>P.</given-names></name></person-group> (<year>2025</year>). <article-title>Dynamical decoupling of generalization and overfitting in large two-layer networks</article-title>. <source>arXiv preprint arXiv:2502.21269</source>.</citation>
</ref>
<ref id="B49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mori</surname> <given-names>F.</given-names></name> <name><surname>Mannelli</surname> <given-names>S. S.</given-names></name> <name><surname>Mignacco</surname> <given-names>F.</given-names></name></person-group> (<year>2025</year>). <article-title>&#x0201C;Optimal protocols for continual learning via statistical physics and control theory,&#x0201D;</article-title> in <source>The Thirteenth International Conference on Learning Representations</source>.</citation>
</ref>
<ref id="B50">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Murthy</surname> <given-names>V. N.</given-names></name></person-group> (<year>2011</year>). <article-title>Olfactory maps in the brain</article-title>. <source>Annu. Rev. Neurosci</source>. <volume>34</volume>, <fpage>233</fpage>&#x02013;<lpage>258</lpage>. <pub-id pub-id-type="doi">10.1146/annurev-neuro-061010-113738</pub-id><pub-id pub-id-type="pmid">21692659</pub-id></citation></ref>
<ref id="B51">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>N&#x000F8;kland</surname> <given-names>A.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Direct feedback alignment provides learning in deep neural networks,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source>, 29.</citation>
</ref>
<ref id="B52">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pashakhanloo</surname> <given-names>F.</given-names></name> <name><surname>Koulakov</surname> <given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Stochastic gradient descent-induced drift of representation in a two-layer neural network,&#x0201D;</article-title> in <source>Proceedings of the 40th International Conference on Machine Learning, volume 202 of Proceedings of Machine Learning Research</source>, eds. A. Krause, E. Brunskill, K. Cho, B. Engelhardt, S. Sabato, and J. Scarlett (PMLR), <fpage>27401</fpage>&#x02013;<lpage>27419</lpage>.</citation>
</ref>
<ref id="B53">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qin</surname> <given-names>S.</given-names></name> <name><surname>Farashahi</surname> <given-names>S.</given-names></name> <name><surname>Lipshutz</surname> <given-names>D.</given-names></name> <name><surname>Sengupta</surname> <given-names>A. M.</given-names></name> <name><surname>Chklovskii</surname> <given-names>D. B.</given-names></name> <name><surname>Pehlevan</surname> <given-names>C.</given-names></name></person-group> (<year>2023</year>). <article-title>Coordinated drift of receptive fields in Hebbian/anti-Hebbian network models during noisy representation learning</article-title>. <source>Nat. Neurosci</source>. <volume>26</volume>, <fpage>339</fpage>&#x02013;<lpage>349</lpage>. <pub-id pub-id-type="doi">10.1038/s41593-022-01225-z</pub-id><pub-id pub-id-type="pmid">36635497</pub-id></citation></ref>
<ref id="B54">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rule</surname> <given-names>M. E.</given-names></name> <name><surname>O&#x00027;Leary</surname> <given-names>T.</given-names></name> <name><surname>Harvey</surname> <given-names>C. D.</given-names></name></person-group> (<year>2019</year>). <article-title>Causes and consequences of representational drift</article-title>. <source>Curr. Opin. Neurobiol</source>. <volume>58</volume>, <fpage>141</fpage>&#x02013;<lpage>147</lpage>. <pub-id pub-id-type="doi">10.1016/j.conb.2019.08.005</pub-id><pub-id pub-id-type="pmid">31569062</pub-id></citation></ref>
<ref id="B55">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Saad</surname> <given-names>D.</given-names></name> <name><surname>Solla</surname> <given-names>S. A.</given-names></name></person-group> (<year>1995</year>). <article-title>On-line learning in soft committee machines</article-title>. <source>Phys. Rev. E</source> <volume>52</volume>:<fpage>4225</fpage>. <pub-id pub-id-type="doi">10.1103/PhysRevE.52.4225</pub-id></citation>
</ref>
<ref id="B56">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Saxe</surname> <given-names>A. M.</given-names></name> <name><surname>McClelland</surname> <given-names>J. L.</given-names></name> <name><surname>Ganguli</surname> <given-names>S.</given-names></name></person-group> (<year>2013</year>). <article-title>Exact solutions to the nonlinear dynamics of learning in deep linear neural networks</article-title>. <source>arXiv preprint arXiv:1312.6120</source>.</citation>
</ref>
<ref id="B57">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sorscher</surname> <given-names>B.</given-names></name> <name><surname>Ganguli</surname> <given-names>S.</given-names></name> <name><surname>Sompolinsky</surname> <given-names>H.</given-names></name></person-group> (<year>2022</year>). <article-title>Neural representational geometry underlies few-shot concept learning</article-title>. <source>Proc. Nat. Acad. Sci</source>. <volume>119</volume>:<fpage>e2200800119</fpage>. <pub-id pub-id-type="doi">10.1073/pnas.2200800119</pub-id><pub-id pub-id-type="pmid">36251997</pub-id></citation></ref>
<ref id="B58">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Steinmetz</surname> <given-names>N. A.</given-names></name> <name><surname>Aydin</surname> <given-names>C.</given-names></name> <name><surname>Lebedeva</surname> <given-names>A.</given-names></name> <name><surname>Okun</surname> <given-names>M.</given-names></name> <name><surname>Pachitariu</surname> <given-names>M.</given-names></name> <name><surname>Bauza</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Neuropixels 2.0: a miniaturized high-density probe for stable, long-term brain recordings</article-title>. <source>Science</source> <volume>372</volume>:<fpage>eabf4588</fpage>. <pub-id pub-id-type="doi">10.1126/science.abf4588</pub-id><pub-id pub-id-type="pmid">33859006</pub-id></citation></ref>
<ref id="B59">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stiso</surname> <given-names>J.</given-names></name> <name><surname>Bassett</surname> <given-names>D. S.</given-names></name></person-group> (<year>2018</year>). <article-title>Spatial embedding imposes constraints on neuronal network architectures</article-title>. <source>Trends Cogn. Sci</source>. <volume>22</volume>, <fpage>1127</fpage>&#x02013;<lpage>1142</lpage>. <pub-id pub-id-type="doi">10.1016/j.tics.2018.09.007</pub-id><pub-id pub-id-type="pmid">30449318</pub-id></citation></ref>
<ref id="B60">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sucholutsky</surname> <given-names>I.</given-names></name> <name><surname>Muttenthaler</surname> <given-names>L.</given-names></name> <name><surname>Weller</surname> <given-names>A.</given-names></name> <name><surname>Peng</surname> <given-names>A.</given-names></name> <name><surname>Bobu</surname> <given-names>A.</given-names></name> <name><surname>Kim</surname> <given-names>B.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Getting aligned on representational alignment</article-title>. <source>arXiv preprint arXiv:2310.13018</source>.</citation>
</ref>
<ref id="B61">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>W.</given-names></name> <name><surname>Winnubst</surname> <given-names>J.</given-names></name> <name><surname>Natrajan</surname> <given-names>M.</given-names></name> <name><surname>Lai</surname> <given-names>C.</given-names></name> <name><surname>Kajikawa</surname> <given-names>K.</given-names></name> <name><surname>Bast</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>Learning produces an orthogonalized state machine in the hippocampus</article-title>. <source>Nature</source> <volume>640</volume>, <fpage>165</fpage>&#x02013;<lpage>175</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-024-08548-w</pub-id><pub-id pub-id-type="pmid">39939774</pub-id></citation></ref>
<ref id="B62">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vaidya</surname> <given-names>S. P.</given-names></name> <name><surname>Li</surname> <given-names>G.</given-names></name> <name><surname>Chitwood</surname> <given-names>R. A.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Magee</surname> <given-names>J. C.</given-names></name></person-group> (<year>2025</year>). <article-title>Formation of an expanding memory representation in the hippocampus</article-title>. <source>Nat. Neurosci</source>. <volume>28</volume>, <fpage>1510</fpage>&#x02013;<lpage>1518</lpage>. <pub-id pub-id-type="doi">10.1038/s41593-025-01986-3</pub-id><pub-id pub-id-type="pmid">40467863</pub-id></citation></ref>
<ref id="B63">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>van Meegen</surname> <given-names>A.</given-names></name> <name><surname>Sompolinsky</surname> <given-names>H.</given-names></name></person-group> (<year>2025</year>). <article-title>Coding schemes in neural networks learning classification tasks</article-title>. <source>Nat. Commun</source>. <volume>16</volume>:<fpage>3354</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-025-58276-6</pub-id><pub-id pub-id-type="pmid">40204730</pub-id></citation></ref>
<ref id="B64">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Watkin</surname> <given-names>T. L. H.</given-names></name> <name><surname>Rau</surname> <given-names>A.</given-names></name> <name><surname>Biehl</surname> <given-names>M.</given-names></name></person-group> (<year>1993</year>). <article-title>The statistical mechanics of learning a rule</article-title>. <source>Rev. Mod. Phys</source>. <volume>65</volume>, <fpage>499</fpage>&#x02013;<lpage>556</lpage>. <pub-id pub-id-type="doi">10.1103/RevModPhys.65.499</pub-id></citation>
</ref>
<ref id="B65">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Williams</surname> <given-names>A. H.</given-names></name></person-group> (<year>2024</year>). <article-title>Equivalence between representational similarity analysis, centered kernel alignment, and canonical correlations analysis</article-title>. <source>bioRxiv. 2024-10</source>. <pub-id pub-id-type="doi">10.1101/2024.10.23.619871</pub-id></citation>
</ref>
<ref id="B66">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Williams</surname> <given-names>A. H.</given-names></name> <name><surname>Kunz</surname> <given-names>E.</given-names></name> <name><surname>Kornblith</surname> <given-names>S.</given-names></name> <name><surname>Linderman</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Generalized shape metrics on neural representations,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source>, eds. M. Ranzato, A. Beygelzimer, Y. Dauphin, P. Liang, and J. W. Vaughan (Curran Associates, Inc.), <fpage>4738</fpage>&#x02013;<lpage>4750</lpage>.</citation>
</ref>
<ref id="B67">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Williams</surname> <given-names>C.</given-names></name></person-group> (<year>1996</year>). <article-title>&#x0201C;Computing with infinite networks,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source>, eds. M. Mozer, M. Jordan, and T. Petsche (MIT Press).</citation>
</ref>
<ref id="B68">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>G.</given-names></name> <name><surname>Hu</surname> <given-names>E. J.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Tensor programs IV: feature learning in infinite-width neural networks,&#x0201D;</article-title> in <source>International Conference on Machine Learning</source> (<publisher-loc>PMLR</publisher-loc>), <fpage>11727</fpage>&#x02013;<lpage>11737</lpage>.</citation>
</ref>
<ref id="B69">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zavatone-Veth</surname> <given-names>J. A.</given-names></name> <name><surname>Canatar</surname> <given-names>A.</given-names></name> <name><surname>Ruben</surname> <given-names>B. S.</given-names></name> <name><surname>Pehlevan</surname> <given-names>C.</given-names></name></person-group> (<year>2022a</year>). <article-title>Asymptotics of representation learning in finite Bayesian neural networks</article-title>. <source>J. Statistical Mech</source>. <volume>2022</volume>:<fpage>114008</fpage>. <pub-id pub-id-type="doi">10.1088/1742-5468/ac98a6</pub-id></citation>
</ref>
<ref id="B70">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zavatone-Veth</surname> <given-names>J. A.</given-names></name> <name><surname>Pehlevan</surname> <given-names>C.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Depth induces scale-averaging in overparameterized linear Bayesian neural networks,&#x0201D;</article-title> in <source>Asilomar Conference on Signals, Systems, and Computers</source>, 55. <pub-id pub-id-type="doi">10.1109/IEEECONF53345.2021.9723137</pub-id></citation>
</ref>
<ref id="B71">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zavatone-Veth</surname> <given-names>J. A.</given-names></name> <name><surname>Pehlevan</surname> <given-names>C.</given-names></name></person-group> (<year>2022</year>). <article-title>On neural network kernels and the storage capacity problem</article-title>. <source>Neural Comput</source>. <volume>34</volume>, <fpage>1136</fpage>&#x02013;<lpage>1142</lpage>. <pub-id pub-id-type="doi">10.1162/neco_a_01494</pub-id><pub-id pub-id-type="pmid">35344992</pub-id></citation></ref>
<ref id="B72">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zavatone-Veth</surname> <given-names>J. A.</given-names></name> <name><surname>Tong</surname> <given-names>W. L.</given-names></name> <name><surname>Pehlevan</surname> <given-names>C.</given-names></name></person-group> (<year>2022b</year>). <article-title>Contrasting random and learned features in deep Bayesian linear regression</article-title>. <source>Phys. Rev. E</source> <volume>105</volume>:<fpage>064118</fpage>. <pub-id pub-id-type="doi">10.1103/PhysRevE.105.064118</pub-id><pub-id pub-id-type="pmid">35854590</pub-id></citation></ref>
<ref id="B73">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zdeborov&#x000E1;</surname> <given-names>L.</given-names></name> <name><surname>Krzakala</surname> <given-names>F.</given-names></name></person-group> (<year>2016</year>). <article-title>Statistical physics of inference: thresholds and algorithms</article-title>. <source>Adv. Phys</source>. <volume>65</volume>, <fpage>453</fpage>&#x02013;<lpage>552</lpage>. <pub-id pub-id-type="doi">10.1080/00018732.2016.1211393</pub-id></citation>
</ref>
<ref id="B74">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>M.</given-names></name> <name><surname>Pan</surname> <given-names>X.</given-names></name> <name><surname>Jung</surname> <given-names>W.</given-names></name> <name><surname>Halpern</surname> <given-names>A. R.</given-names></name> <name><surname>Eichhorn</surname> <given-names>S. W.</given-names></name> <name><surname>Lei</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Molecularly defined and spatially resolved cell atlas of the whole mouse brain</article-title>. <source>Nature</source> <volume>624</volume>, <fpage>343</fpage>&#x02013;<lpage>354</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-023-06808-9</pub-id><pub-id pub-id-type="pmid">38092912</pub-id></citation></ref>
<ref id="B75">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhong</surname> <given-names>L.</given-names></name> <name><surname>Baptista</surname> <given-names>S.</given-names></name> <name><surname>Gattoni</surname> <given-names>R.</given-names></name> <name><surname>Arnold</surname> <given-names>J.</given-names></name> <name><surname>Flickinger</surname> <given-names>D.</given-names></name> <name><surname>Stringer</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>Unsupervised pretraining in biological neural networks</article-title>. <source>Nature</source> <volume>2025</volume>, <fpage>1</fpage>&#x02013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-025-09180-y</pub-id><pub-id pub-id-type="pmid">40533561</pub-id></citation></ref>
</ref-list>
</back>
</article>