<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Signal Process.</journal-id>
<journal-title>Frontiers in Signal Processing</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Signal Process.</abbrev-journal-title>
<issn pub-type="epub">2673-8198</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1522604</article-id>
<article-id pub-id-type="doi">10.3389/frsip.2025.1522604</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Signal Processing</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Evaluating CPU, GPU, and FPGA performance in the context of modal reverberation: a comparative analysis</article-title>
<alt-title alt-title-type="left-running-head">Michon et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/frsip.2025.1522604">10.3389/frsip.2025.1522604</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Michon</surname>
<given-names>Romain</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2701869/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ducceschi</surname>
<given-names>Michele</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2510689/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cochard</surname>
<given-names>Pierre</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Skare</surname>
<given-names>Travis</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2886900/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Webb</surname>
<given-names>Craig J.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Russo</surname>
<given-names>Riccardo</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>INRIA</institution>, <institution>CITI Laboratory</institution>, <institution>INSA Lyon</institution>, <institution>GRAME-CNCM</institution>, <addr-line>Villeurbanne</addr-line>, <country>France</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>NEMUS Lab</institution>, <institution>Department of Industrial Engineering</institution>, <institution>University of Bologna</institution>, <addr-line>Bologna</addr-line>, <country>Italy</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Electrical Engineering</institution>, <institution>Stanford University</institution>, <addr-line>Stanford</addr-line>, <addr-line>CA</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2048350/overview">Antonio Peto&#x161;i&#x107;</ext-link>, University of Zagreb, Croatia</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2190269/overview">Aleksandr Cariow</ext-link>, West Pomeranian University of Technology, Poland</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2897311/overview">Jose Milet Rodriguez Borbon</ext-link>, University of California, Riverside, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Romain Michon, <email>romain.michon@inria.fr</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>04</day>
<month>04</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>5</volume>
<elocation-id>1522604</elocation-id>
<history>
<date date-type="received">
<day>04</day>
<month>11</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>10</day>
<month>03</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Michon, Ducceschi, Cochard, Skare, Webb and Russo.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Michon, Ducceschi, Cochard, Skare, Webb and Russo</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>The vibration of acoustic systems can be represented through modal decomposition, reducing the problem to a set of harmonic oscillators. This study investigates the real-time performance of CPUs, GPUs, and FPGAs in implementing such models, focusing on the synthesis of large-scale modal reverberation. By leveraging their respective architectures, these processors are assessed for their ability to manage the high computational demands of modal synthesis. GPUs and FPGAs, known for their parallel processing capabilities, are evaluated alongside recent multi-core CPUs, which increasingly approach similar performance levels in handling such tasks. Through a series of platform-specific optimisations, this paper examines the maximum achievable mode count, latency, and processing efficiency for each platform in various real-time scenarios. Results indicate that while GPUs offer superior scalability, FPGAs achieve unparalleled latency performance, making them suitable for specific low-latency applications. CPUs, conversely, demonstrate unexpectedly high performance in smaller-scale applications. This work provides insight into the practical application of each processor type within real-time digital signal processing and suggests pathways for future research in hardware-based audio DSP.</p>
</abstract>
<kwd-group>
<kwd>modal reverberation</kwd>
<kwd>parallel computing</kwd>
<kwd>FPGA</kwd>
<kwd>GPU</kwd>
<kwd>CPU</kwd>
<kwd>physical modelling</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Audio and Acoustic Signal Processing</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Modal methods are a foundational technique in computer-aided sound synthesis, representing some of the earliest examples of real-time physical models altogether. Modal projections, akin to Galerkin-like and spectral techniques (<xref ref-type="bibr" rid="B9">Boyd, 2001</xref>; <xref ref-type="bibr" rid="B28">Meirovitch, 2010</xref>), began in earnest as a viable digital synthesis technique with the works by J.M. Adrien and associates at IRCAM (<xref ref-type="bibr" rid="B2">Adrien, 1991</xref>) and with projects such as Mosaic and Modalys (<xref ref-type="bibr" rid="B20">Eckel, 1995</xref>). Such early success was partly due to the naturally parallel structure of the modal decomposition. In linear, time-invariant systems, multiple independent solutions coexist, each contributing to a system&#x2019;s global response to a given input. Such physical independence can be naturally exploited at the computational level through parallelisation. In a typical workflow, the input is first projected onto the modes, which are then updated in parallel, and their contribution is summed, yielding the global system&#x2019;s response. Modal techniques form the core current synthesis techniques, and share common features with other synthesis methods such as the Udwadia-Kalaba method (<xref ref-type="bibr" rid="B14">Debut and Antunes, 2020</xref>), the Functional Transformation Method (<xref ref-type="bibr" rid="B34">Rabenstein and Trautmann, 2003</xref>), and current machine-learning approaches (<xref ref-type="bibr" rid="B41">Schlecht et al., 2022</xref>). They have been extended to treat typical nonlinearities arising in physical models, such as contact friction nonlinearities (<xref ref-type="bibr" rid="B38">Russo et al., 2022</xref>) and intermittent contact (<xref ref-type="bibr" rid="B47">van Walstijn et al., 2016</xref>; <xref ref-type="bibr" rid="B18">Ducceschi et al., 2023</xref>).</p>
<p>One common area of application, and one which has gained much prominence in recent years, is modal reverberation. In this case, the simulated system is not a musical instrument <italic>per se</italic> but a large resonator fed with incoming dry audio. These systems are characterised by a considerable modal density, with thousands of overlapping modes contributing to forming the system&#x2019;s response. Rooms and acoustic enclosures are prominent cases (<xref ref-type="bibr" rid="B31">Pierce, 2019</xref>). Plates and springs, introduced initially as means to sustain sound through more portable devices, are further examples (<xref ref-type="bibr" rid="B44">Valimaki et al., 2012</xref>). In the context of artificial reverberation, an additional distinction can be made based on the specific method used to derive the modal parameters&#x2014;whether it is model-based or signal-based (<xref ref-type="bibr" rid="B1">Abel et al., 2014</xref>). In the former, the modes are readily derived from an input model of the target system, such as springs (<xref ref-type="bibr" rid="B27">McQuillan and van Walstijn, 2021</xref>; <xref ref-type="bibr" rid="B46">van Walstijn, 2020</xref>) and plates (<xref ref-type="bibr" rid="B19">Ducceschi and Webb, 2016</xref>). In signal-based modal synthesis, the modal parameters are first identified through a suitable algorithm either in the time or in the frequency domain (<xref ref-type="bibr" rid="B4">Avitabile, 2017</xref>), and the resulting system&#x2019;s response is then synthesised via a parallel biquad filter structure. Examples of such applications abound, particularly in room acoustics, where the modal identification is carried out both in the time domain (<xref ref-type="bibr" rid="B22">Kereliuk et al., 2018</xref>; <xref ref-type="bibr" rid="B35">Rau et al., 2021</xref>) using modifications of the ESPRIT algorithm (<xref ref-type="bibr" rid="B37">Roy and Kailath, 1989</xref>), as well as in the frequency domain via nonlinear optimisation (<xref ref-type="bibr" rid="B25">Maestre et al., 2017</xref>; <xref ref-type="bibr" rid="B26">2016</xref>; <xref ref-type="bibr" rid="B6">Bank and Ramos, 2011</xref>).</p>
<p>Due to their parallel nature, modal synthesis algorithms present a promising approach for certain processors that offer high parallel computing capabilities, such as Field-Programmable Gate Arrays (FPGAs) and Graphics Processing Units (GPUs). This paper aims to explore the feasibility of executing modal physical modelling algorithms in real-time on these platforms as a potential alternative to Central Processing Units (CPUs). Specifically, a metal plate with simply-supported boundaries is considered, for which the modal parameters can be directly obtained in closed-form from a model PDE (Partial Differential Equation) (<xref ref-type="bibr" rid="B19">Ducceschi and Webb, 2016</xref>; <xref ref-type="bibr" rid="B49">Willemsen et al., 2017</xref>; <xref ref-type="bibr" rid="B39">Russo et al., 2023</xref>). This algorithm is used to synthesise the maximum possible number of modes across various CPUs, FPGAs, and GPUs, involving a wide range of platform-specific optimisations, which are detailed herein. The findings indicate that the latest generations of CPUs increasingly exhibit performance levels competitive with those of high-end FPGAs and GPUs.</p>
<p>The structure of the paper is as follows: an overview of previous research on executing real-time audio DSP algorithms on GPUs and FPGAs is provided initially. This is followed by a presentation of the modal synthesis algorithm, along with a comprehensive description of the platform-specific optimisations implemented for various CPUs, FPGAs, and GPUs. The performance of this algorithm on each platform is then reported, leading to a discussion of these results and suggestions for potential future research directions.</p>
</sec>
<sec id="s2">
<title>2 Real-time audio DSP on GPUs and FPGAs</title>
<p>This section provides an overview of GPU and FPGA usage in real-time audio DSP, with an emphasis on key historical developments in both platforms and their applications in audio processing and synthesis.</p>
<sec id="s2-1">
<title>2.1 GPUs</title>
<p>In recent years, GPUs have been extensively employed in machine learning training and real-time inference applications. Real-time audio synthesis and effects processing on GPUs, however, remains a niche area of research, with notable progress emerging since the introduction of CUDA and unified shader pipelines in 2006. Early work in this domain, such as <xref ref-type="bibr" rid="B40">Savioja et al. (2010)</xref>, demonstrated the feasibility of synthesising one million sinusoids at audio rates, albeit through computing multiple adjacent time samples in parallel to accommodate the GPU clock rates available at the time.</p>
<p>With hardware advances, running certain audio synthesis and filtering tasks with inter-sample dependencies became possible. <xref ref-type="bibr" rid="B36">Renney et al. (2020)</xref> test real-time feasibility of more modern Nvidia and AMD GPUs. Work such as the NESS project (<xref ref-type="bibr" rid="B8">Bilbao et al., 2019</xref>) utilised GPUs for high-quality, physically accurate audio synthesis, though generally slower than real-time. There have been commercial products released to end users utilizing commodity GPUs. GPU Audio, Inc.<xref ref-type="fn" rid="fn1">
<sup>1</sup>
</xref> has developed a SDK for partners to integrate or develop GPU-based plugins. Some features such as parallel GPU-accelerated convolution in the <italic>Vienna Symphonic Library</italic> plugin are available to end-users. Several years prior, CUDA acceleration was availble in Acusticaudio s. r.l&#x2019;s Nebula 3,<xref ref-type="fn" rid="fn2">
<sup>2</sup>
</xref> though this functionality was later removed. Finally, in prior work, <xref ref-type="bibr" rid="B42">Skare and Abel (2019)</xref> demonstrated the ability to run a modal filter bank based on phasor filters at audio rates that could accept MIDI input to synthesise percussion sounds.</p>
</sec>
<sec id="s2-2">
<title>2.2 FPGAs</title>
<p>FPGAs have gained prominence as a platform for real-time audio Digital Signal Processing (DSP) within both industry and academic environments. Their low-level architecture enables them to achieve unmatched real-time performance in terms of audio latency and exceptionally high sampling rates (<xref ref-type="bibr" rid="B33">Popoff et al., 2022</xref>). The reconfigurable nature of FPGAs allows for maximised optimisation tailored to specific applications, with adaptable levels of parallelisation and pipelining to attain the optimal performance that the system can support.</p>
<p>Early contributions to audio processing on Field-Programmable Gate Arrays (FPGAs) emerged in the 2000s, primarily focusing on the manual implementation of specific applications or algorithms using Hardware Description Languages (HDLs) such as VHDL or Verilog on FPGA-only boards. For instance, <xref ref-type="bibr" rid="B29">Motuk et al. (2007)</xref> provides a notable example, aiming to implement Finite-Difference Time-Domain (FDTD) models. Similarly, other projects have concentrated on digital drum kits (<xref ref-type="bibr" rid="B21">Jadhao and Singh Patel, 2020</xref>), audio effect generators (<xref ref-type="bibr" rid="B11">Chhetri et al., 2015</xref>; <xref ref-type="bibr" rid="B17">Dragoi et al., 2021</xref>), among others. More recent developments, with the integration of System on Chip (SoC)<xref ref-type="fn" rid="fn3">
<sup>3</sup>
</xref> capabilities in contemporary FPGA platforms, have facilitated fully standalone applications capable of managing both audio control and processing, thus enabling a hardware<xref ref-type="fn" rid="fn4">
<sup>4</sup>
</xref>/software co-design approach (<xref ref-type="bibr" rid="B10">Cannon et al., 2022</xref>; <xref ref-type="bibr" rid="B16">Deulkar and Kolhare, 2020</xref>; <xref ref-type="bibr" rid="B43">Vaca et al., 2019</xref>). Within industry, companies such as Novation,<xref ref-type="fn" rid="fn5">
<sup>5</sup>
</xref> Antelope Audio,<xref ref-type="fn" rid="fn6">
<sup>6</sup>
</xref> UDO Audio,<xref ref-type="fn" rid="fn7">
<sup>7</sup>
</xref> futur3soundz,<xref ref-type="fn" rid="fn8">
<sup>8</sup>
</xref> and Audinate<xref ref-type="fn" rid="fn9">
<sup>9</sup>
</xref> offer products that incorporate FPGA technology.</p>
<p>When it comes to implementing audio DSP algorithms on an FPGA, various approaches can be taken. As mentioned above, the most common and &#x201c;obvious&#x201d; one is to write HDL code &#x201c;by hand.&#x201d; This solution is only accessible to hardware engineers with highly specialised skills and is therefore out of reach to most DSP and software engineers. Environments such as Matlab Simulink<xref ref-type="fn" rid="fn10">
<sup>10</sup>
</xref> enables the assembly of pre-designed DSP blocks to implement specific applications. While this solution is good for basic prototyping, it remains limited, especially when it comes to using custom algorithms. MathWorks&#x2019; HDL Coder<xref ref-type="fn" rid="fn11">
<sup>11</sup>
</xref> is another solution which allows for the programming of FPGAs at a high-level using Matlab. While it has been successfully used in a couple of projects in academia for real-time audio DSP applications (<xref ref-type="bibr" rid="B45">Vannoy, 2020</xref>), it mostly targets rapid prototyping to the detriment of optimisation and computational efficiency. <xref ref-type="bibr" rid="B48">Verstraelen et al. (2014)</xref> proposed a programmable parallel machine on FPGA targeting audio applications, but this project is not active anymore.</p>
<p>More recently, High-Level Synthesis (HLS) (<xref ref-type="bibr" rid="B24">Lahti et al., 2018</xref>) has proven to balance ease of programming and performance. In this context, the open source Syfala project,<xref ref-type="fn" rid="fn12">
<sup>12</sup>
</xref> which relies on the vitis_hls tool provided by Xilinx/AMD, has been aiming at providing an optimised &#x201c;audio DSP to FPGA compiler&#x201d; taking both C/C&#x2b;&#x2b; or Faust (<xref ref-type="bibr" rid="B30">Orlarey et al., 2009</xref>) code as an input (<xref ref-type="bibr" rid="B33">Popoff et al., 2022</xref>). Syfala can target various Xilinx/AMD FPGA boards and provide a broad range of side features such as various sister boards for control and multichannel audio applications (<xref ref-type="bibr" rid="B32">Popoff et al., 2024</xref>), Open Sound Control (OSC), MIDI, etc. control, hardware acceleration on dedicated Linux (<xref ref-type="bibr" rid="B13">Cochard et al., 2024</xref>), etc. When using C/C&#x2b;&#x2b;, Syfala heavily relies on HLS pragmas for optimisation and code must respect various standards in order to be as efficient as possible.<xref ref-type="fn" rid="fn13">
<sup>13</sup>
</xref>
</p>
</sec>
</sec>
<sec id="s3">
<title>3 Modal synthesis and reverberation</title>
<p>Before proceeding, it is worth recalling the mathematics of modal reverberation. The purpose of this section is to describe the modal algorithms and introduce notation. As mentioned in the introduction, in model-based modal reverberation, the modal parameters are either assumed to be known (<xref ref-type="bibr" rid="B19">Ducceschi and Webb, 2016</xref>; <xref ref-type="bibr" rid="B39">Russo et al., 2023</xref>) or are obtained through a numerical eigenvalue problem (<xref ref-type="bibr" rid="B46">van Walstijn, 2020</xref>; <xref ref-type="bibr" rid="B27">McQuillan and van Walstijn, 2021</xref>). Here, as the focus is on the performance of modal algorithms, the former approach is chosen for a thin metallic plate with a simply-supported boundary. This serves as a first approximation to plate reverb units such as the EMT140 (<xref ref-type="bibr" rid="B3">Arcas and Chaigne, 2010</xref>; <xref ref-type="bibr" rid="B44">Valimaki et al., 2012</xref>). To that end, consider the following model describing the flexural vibration of a thin plate (<xref ref-type="bibr" rid="B7">Bilbao et al., 2006</xref>):<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mi>u</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:mi>u</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3ba;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:mi>u</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>u</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>In the above, <inline-formula id="inf1">
<mml:math id="m2">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>:</mml:mo>
<mml:mi mathvariant="script">D</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2b;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2192;</mml:mo>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the flexural displacement of the metallic sheet over a rectangular domain <inline-formula id="inf2">
<mml:math id="m3">
<mml:mrow>
<mml:mi mathvariant="script">D</mml:mi>
<mml:mo>&#x2254;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf3">
<mml:math id="m4">
<mml:mrow>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf4">
<mml:math id="m5">
<mml:mrow>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are, respectively, the Laplace and the biharmonic operators. In Cartesian coordinates, these are:<disp-formula id="equ1">
<mml:math id="m6">
<mml:mrow>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:mo>&#x2254;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:mo>&#x2254;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>Furthermore, in (<xref ref-type="disp-formula" rid="e1">Equation 1</xref>), <inline-formula id="inf5">
<mml:math id="m7">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x2254;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mi>&#x3c1;</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is a tension term, with <inline-formula id="inf6">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> being the applied tension per unit length along the edges, <inline-formula id="inf7">
<mml:math id="m9">
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> being the metal density and <inline-formula id="inf8">
<mml:math id="m10">
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> being the thickness of the plate (of the order of half a millimetre for steel reverb units). <inline-formula id="inf9">
<mml:math id="m11">
<mml:mrow>
<mml:mi>&#x3ba;</mml:mi>
<mml:mo>&#x2254;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>/</mml:mo>
<mml:mi>&#x3c1;</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is a stiffness constant, with <inline-formula id="inf10">
<mml:math id="m12">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>&#x2254;</mml:mo>
<mml:mi>E</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>/</mml:mo>
<mml:mn>12</mml:mn>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bd;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> being a rigidity constant, <inline-formula id="inf11">
<mml:math id="m13">
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> being Young&#x2019;s modulus and <inline-formula id="inf12">
<mml:math id="m14">
<mml:mrow>
<mml:mi>&#x3bd;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> being Poisson&#x2019;s ratio. The constant <inline-formula id="inf13">
<mml:math id="m15">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a loss factor, <inline-formula id="inf14">
<mml:math id="m16">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the incoming dry audio, and <inline-formula id="inf15">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the input location on the plate.</p>
<p>A particular solution to (<xref ref-type="disp-formula" rid="e1">Equation 1</xref>) is obtained by first solving the eigenvalue problem defined on the lossless system, that is:<disp-formula id="e2">
<mml:math id="m18">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mi>U</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:mi>U</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3ba;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:mi>U</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>together with the boundary conditions <inline-formula id="inf16">
<mml:math id="m19">
<mml:mrow>
<mml:mi>U</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:mi>U</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> along the boundary <inline-formula id="inf17">
<mml:math id="m20">
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi mathvariant="script">D</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, for a modal shape <inline-formula id="inf18">
<mml:math id="m21">
<mml:mrow>
<mml:mi>U</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. A particular solution is given by:<disp-formula id="e3">
<mml:math id="m22">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2254;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
<mml:mi>sin</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
<mml:mi>sin</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>such that (<xref ref-type="disp-formula" rid="e2">Equation 2</xref>) is solved by:<disp-formula id="equ2">
<mml:math id="m23">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2254;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3ba;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</disp-formula>for real-valued modal wavenumbers <inline-formula id="inf19">
<mml:math id="m24">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2254;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
</inline-formula> and positive modal indices <inline-formula id="inf20">
<mml:math id="m25">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="double-struck">N</mml:mi>
</mml:math>
</inline-formula>. One may then sort the particular solutions according to increasing frequency, using a single sorting index <inline-formula id="inf21">
<mml:math id="m26">
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Thus, <inline-formula id="inf22">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf23">
<mml:math id="m28">
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the total amount of modes retained in the model. The associated mode shapes are <inline-formula id="inf24">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, as per (<xref ref-type="disp-formula" rid="e3">Equation 3</xref>). The modal equation for the <inline-formula id="inf25">
<mml:math id="m30">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>th</mml:mtext>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> mode is then:<disp-formula id="e4">
<mml:math id="m31">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mo>&#x308;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mo>&#x307;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>where the loss factor <inline-formula id="inf26">
<mml:math id="m32">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is now mode-dependent and can be given in terms of the 60&#xa0;dB&#xa0;decay time <inline-formula id="inf27">
<mml:math id="m33">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> as <inline-formula id="inf28">
<mml:math id="m34">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2254;</mml:mo>
<mml:mn>3</mml:mn>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf29">
<mml:math id="m35">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the time-dependent modal coordinate for mode <inline-formula id="inf30">
<mml:math id="m36">
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Note that, in (<xref ref-type="disp-formula" rid="e4">Equation 4</xref>), total time derivates are now denoted with overdots.</p>
<p>The global solution at location <inline-formula id="inf31">
<mml:math id="m37">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2254;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is then reconstructed as:<disp-formula id="e5">
<mml:math id="m38">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>A schematic representation of the typical workflow in model-based modal synthesis is offered in <xref ref-type="fig" rid="F1">Figure 1</xref>. Pictures of the first few modal shapes are given in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Workflow of model-based modal synthesis. An incoming dry audio signal is projected onto <inline-formula id="inf32">
<mml:math id="m39">
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> modal shapes, yielding the input modal weights. The modal equations are in the form of a bank of <inline-formula id="inf33">
<mml:math id="m40">
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> parallel oscillators, and the global output is then reconstructed through a reduced sum.</p>
</caption>
<graphic xlink:href="frsip-05-1522604-g001.tif"/>
</fig>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The first four modal shapes <inline-formula id="inf34">
<mml:math id="m41">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> for a rectangular plate with <inline-formula id="inf35">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf36">
<mml:math id="m43">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, typical dimensions for plate reverb units such as the EMT140. The shapes are sorted according to increasing frequency, and modal indices <inline-formula id="inf37">
<mml:math id="m44">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are as given.</p>
</caption>
<graphic xlink:href="frsip-05-1522604-g002.tif"/>
</fig>
<p>Before proceeding, it is worth solving (<xref ref-type="disp-formula" rid="e4">Equation 4</xref>) in the frequency domain, after substituting <inline-formula id="inf38">
<mml:math id="m45">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf39">
<mml:math id="m46">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> for complex amplitudes <inline-formula id="inf40">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf41">
<mml:math id="m48">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>. This gives the transfer function:<disp-formula id="e6">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2254;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>2</mml:mn>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<sec id="s3-1">
<title>3.1 Discrete-time update</title>
<p>A time-stepping algorithm is constructed by approximating the continuous solution <inline-formula id="inf42">
<mml:math id="m50">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> with a time series <inline-formula id="inf43">
<mml:math id="m51">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, defined at <inline-formula id="inf44">
<mml:math id="m52">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2254;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf45">
<mml:math id="m53">
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the sampling interval (i.e., the multiplicative inverse of the sample rate), and <inline-formula id="inf46">
<mml:math id="m54">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="double-struck">N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the sampling index. A discrete-time transfer function, discretising <inline-formula id="inf47">
<mml:math id="m55">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> in (<xref ref-type="disp-formula" rid="e8">Equation 8</xref>), is usually obtained using two main approaches. The first is through the integration of (<xref ref-type="disp-formula" rid="e4">Equation 4</xref>) using a basic St&#xf6;rmer-Verlet algorithm as done, e.g., in <xref ref-type="bibr" rid="B19">Ducceschi and Webb (2016)</xref>; the second is via an exact integrator such as the one given in <xref ref-type="bibr" rid="B12">Cie&#x15b;li&#x144;ski (2011)</xref>. The two discrete-time transfer functions follow as.<disp-formula id="e7a">
<mml:math id="m56">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="fraktur">H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2254;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(7a)</label>
</disp-formula>
<disp-formula id="e7b">
<mml:math id="m57">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="fraktur">H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2254;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>cos</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msup>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(7b)</label>
</disp-formula>
</p>
<p>The online computational burden of the two discretisations is the same once the constant coefficients are computed before runtime. (<xref ref-type="disp-formula" rid="e7b">Equation 7b</xref>) is preferred in general as it does not introduce artificial frequency warping. Furthermore, it is unconditionally stable as opposed to (<xref ref-type="disp-formula" rid="e7a">Equation 7a</xref>) for which one must choose <inline-formula id="inf48">
<mml:math id="m58">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>/</mml:mo>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> for stability.</p>
<p>Regardless of the particular form of the transfer function, the prototype algorithm has the form:<disp-formula id="e8">
<mml:math id="m59">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>m</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>M</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>requiring three multiplies and two adds per mode, besides swapping two state arrays after the update. The reduced sum in (<xref ref-type="disp-formula" rid="e5">Equation 5</xref>) is used to compute the output at the output point as:<disp-formula id="e9">
<mml:math id="m60">
<mml:mrow>
<mml:mtext>output</mml:mtext>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mtext>weights</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>&#x2003;</mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
</p>
</sec>
<sec id="s3-2">
<title>3.2 Algorithm architecture</title>
<p>
<xref ref-type="statement" rid="Algorithm_1">Algorithm 1</xref> shows the standard layout of the computation for a modal processor. At each time step, the modal state array uNext is updated using two previous states, u and uPrev, and arrays of coefficients, and the input sample is also added to each modal equation, as per <xref ref-type="disp-formula" rid="e8">Equation 8</xref>. A dot product is subsequently performed over this updated state with an array of weights to generate the output sample for the given time step, as shown in <xref ref-type="disp-formula" rid="e9">Equation 9</xref>. The state arrays are then interchanged prior to initiating the next time iteration.</p>
<p>
<statement content-type="algorithm" id="Algorithm_1">
<label>Algorithm 1</label>
<p>Standard algorithm for Modal Processing.<list list-type="simple">
<list-item>
<p>&#xa0;<bold>for</bold> <inline-formula id="inf49">
<mml:math id="m61">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>: bufferSize-1 <bold>do</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<bold>for</bold> <inline-formula id="inf50">
<mml:math id="m62">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>: numberOfModes-1 <bold>do</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003; <inline-formula id="inf51">
<mml:math id="m63">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003; <inline-formula id="inf511">
<mml:math id="m613">
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mn>3</mml:mn>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="bold">i</mml:mi>
<mml:mi mathvariant="bold">n</mml:mi>
<mml:mi mathvariant="bold">p</mml:mi>
<mml:mi mathvariant="bold">u</mml:mi>
<mml:mi mathvariant="bold">t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003; <bold>end for</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x22b3; Dot product to give output</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003; <inline-formula id="inf52">
<mml:math id="m64">
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>m</mml:mi>
<mml:mo>&#x2190;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<bold>for</bold>&#xa0;<inline-formula id="inf53">
<mml:math id="m65">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>:numberOfModes-1 <bold>do</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003; <inline-formula id="inf54">
<mml:math id="m66">
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>m</mml:mi>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>m</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>w</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<bold>end for</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003; <inline-formula id="inf55">
<mml:math id="m67">
<mml:mrow>
<mml:mi mathvariant="bold">o</mml:mi>
<mml:mi mathvariant="bold">u</mml:mi>
<mml:mi mathvariant="bold">t</mml:mi>
<mml:mi mathvariant="bold">p</mml:mi>
<mml:mi mathvariant="bold">u</mml:mi>
<mml:mi mathvariant="bold">t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x22b3; Swap the state arrays</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003; <inline-formula id="inf56">
<mml:math id="m68">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>v</mml:mi>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003; <inline-formula id="inf57">
<mml:math id="m69">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003; <bold>end</bold>&#xa0;<bold>for</bold>
</p>
</list-item>
</list>
</p>
</statement>
</p>
<p>To produce full-bandwidth audio, the simulation is conducted at 48&#xa0;kHz&#x2014;equivalent to 48,000 time steps, divided into buffers, to yield one second of output. A typical buffer size is 256 samples. The computational cost, therefore, depends on the number of modes employed, which determines the size of the state arrays. Observing that each element of the state array is updated independently, the algorithm can be reformulated as in <xref ref-type="disp-formula" rid="e2">Equation 2</xref>. Rather than looping over time first, the process can loop over the modes of the state, updating each mode across a buffer of time steps. The final output samples are then computed incrementally, and the state of each mode is updated by writing individual values at each time iteration.</p>
<p>
<statement content-type="algorithm" id="Algorithm_2">
<label>Algorithm 2</label>
<p>Modified algorithm with swapped loops.<list list-type="simple">
<list-item>
<p>&#x2003;<bold>for</bold> <inline-formula id="inf58">
<mml:math id="m70">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>:numberOfModes-1&#xa0;<bold>do</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x22b3; Update mode over time-steps</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003; <bold>for</bold> <inline-formula id="inf59">
<mml:math id="m71">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>: bufferSize-1&#xa0;<bold>do</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003; <inline-formula id="inf60">
<mml:math id="m72">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mn>3</mml:mn>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="bold">i</mml:mi>
<mml:mi mathvariant="bold">n</mml:mi>
<mml:mi mathvariant="bold">p</mml:mi>
<mml:mi mathvariant="bold">u</mml:mi>
<mml:mi mathvariant="bold">t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x22b3; Sum into output</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003; <inline-formula id="inf61">
<mml:math id="m73">
<mml:mrow>
<mml:mi mathvariant="bold">o</mml:mi>
<mml:mi mathvariant="bold">u</mml:mi>
<mml:mi mathvariant="bold">t</mml:mi>
<mml:mi mathvariant="bold">p</mml:mi>
<mml:mi mathvariant="bold">u</mml:mi>
<mml:mi mathvariant="bold">t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>w</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> &#x22b3; Swap state of this mode</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003; <inline-formula id="inf62">
<mml:math id="m74">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003; <inline-formula id="inf63">
<mml:math id="m75">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<bold>end</bold>&#xa0;<bold>for</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;<bold>end</bold>&#xa0;<bold>for</bold>
</p>
</list-item>
</list>
</p>
</statement>
</p>
</sec>
</sec>
<sec id="s4">
<title>4 Implementations</title>
<p>The performance and parallelisation methods that these two approaches offer are now discussed.</p>
<sec id="s4-1">
<title>4.1 Central processing unit</title>
<p>In single-threaded CPU execution, the standard algorithm 1) achieves optimal performance when both the state update and dot product are consolidated within a single FOR loop. Compilers such as Clang can fully vectorise this operation under the -Ofast setting. The state arrays may be interchanged through a simple pointer swap, incurring minimal computational cost. In contrast, the modified <xref ref-type="statement" rid="Algorithm_2">algorithm 2</xref>) shows approximately tenfold (x10) slower performance in its unoptimised form. Specifically, the compiler fails to auto-vectorize the FOR loop operations, and each element of the state arrays must be read and written individually to perform the state swap. This memory transfer, coupled with inefficient caching, results in a reduction in overall performance.</p>
<p>To maximise CPU performance, multi-threading is employed to exploit the multi-core architecture. Parallelisation on a limited number of cores is implemented by partitioning the state arrays into discrete sections. For example, with eight threads, each thread processes one-eighth of the modal state, performing the dot product over its designated data segment. Upon completing their respective tasks, the partial sums from the eight threads are aggregated to produce the final output sample. Thread-launch overhead is minimised by allowing each thread to compute a buffer of timesteps, writing the data to a temporary array before synchronising across threads.</p>
<p>This approach, implemented with C&#x2b;&#x2b; stdthreads, is effective across buffer sizes from 128 to 1,024 without impacting performance. Initial offline testing was conducted on a simulation of 48,000 timesteps, corresponding to one second of audio simulation at 48&#xa0;kHz. The primary performance metric was the maximum number of modes that could be computed within a runtime limit of 0.8&#xa0;s, deemed the threshold for usability in a real-time system. Tests were conducted on a range of Apple Silicon machines&#x2014;M1, M2 Pro, and M2 Max processors&#x2014;using configurations of 1, 2, 4, and eight execution threads. The results are presented in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Number of modes computed in an average of 0.8&#xa0;s of run-time on the CPU over 48000 time-steps.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Threads</th>
<th align="center">M1</th>
<th align="center">M2 Pro</th>
<th align="center">M2 Max</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">1</td>
<td align="center">70k</td>
<td align="center">65k</td>
<td align="center">65k</td>
</tr>
<tr>
<td align="left">2</td>
<td align="center">130k</td>
<td align="center">120k</td>
<td align="center">120k</td>
</tr>
<tr>
<td align="left">4</td>
<td align="center">240k</td>
<td align="center">230k</td>
<td align="center">230k</td>
</tr>
<tr>
<td align="left">8</td>
<td align="center">280k</td>
<td align="center">300k</td>
<td align="center">420k</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4-2">
<title>4.2 Graphics processing unit</title>
<p>Although GPUs have long been used for general-purpose computing, the introduction of Nvidia&#x2019;s CUDA platform in 2007 significantly expanded accessibility due to its simplified programming API. Prior to this, GPGPU code was developed using complex shader languages, which presented a steep learning curve. CUDA introduced a straightforward API with C/C&#x2b;&#x2b; extensions, rapidly establishing itself as the default platform for GPU parallel computation code in scientific applications.GPUs are designed on a markedly different model from standard CPUs, which typically contain a limited number of cores, multiple levels of cache, and access to a centralised pool of global memory. In contrast, GPUs consist of thousands of compute cores, organised into Streaming Multiprocessors (SMs) and support multiple memory environments, including local, shared, and global memory. This architecture enables the simultaneous execution of numerous threads, thereby providing substantial parallel computational capabilities. <xref ref-type="fig" rid="F3">Figure 3</xref> illustrates the threading mechanism, wherein blocks of threads are grouped and executed on an SM. <xref ref-type="bibr" rid="B23">Kirk and Wen-Mei (2016)</xref> provides a comprehensive overview of CUDA hardware, programming, and optimisation.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>CUDA thread model. Nvidia Corp.</p>
</caption>
<graphic xlink:href="frsip-05-1522604-g003.tif"/>
</fig>
<p>The effective utilisation of this computational capability depends on the specific algorithm in use. Optimal GPU performance requires issuing thousands of threads, each executing a kernel on data-independent memory, along with an efficient ratio of computation to memory access. To surpass CPU performance, especially in large-scale simulations, millions of threads must be issued to fully harness the GPU&#x2019;s available computational resources. Potential speedup factors and runtime constraints are application-dependent, though studies such as <xref ref-type="bibr" rid="B5">Bakhoda et al. (2009)</xref> comment on real-world performance and optimisation strategies across domains.</p>
<sec id="s4-2-1">
<title>4.2.1 Metrics</title>
<p>Latency and throughput are measured as follows. For a GPU implementation, latency refers to the &#x201c;loop computation&#x201d; time required for the GPU to synthesise up to <inline-formula id="inf64">
<mml:math id="m76">
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> modes, await the completion of this computation, and then sum the results into a mono audio channel. Transfer of input filter parameters to the GPU and transfer of output audio data from the GPU are included in total time. Thus, the total latency corresponds to the wall-clock time necessary to compute one second of audio. In some experiments, audio is processed in chunks with buffer sizes aligned to those requested by real-time digital audio workstation processes, which may also enable adjustments in the scope of sub-problems. For cases involving multiple buffers, median, tail, and maximum latencies are reported, with the stipulation that a real-time system must not exceed the allowed time for processing a callback (sample rate divided by buffer size), allowing for some overhead.</p>
<p>Throughput is determined by the maximum number of modes a system can synthesise to generate one second of audio within 800 milliseconds, with the remaining 200 milliseconds reserved as a buffer. This metric aligns with the previously described CPU implementation.</p>
</sec>
<sec id="s4-2-2">
<title>4.2.2 Implementation</title>
<p>The initial approach for the CUDA implementation was to use the standard algorithm. This requires 3 separate kernels: one to update the state, another to perform the dot product, and then a final thread to load the result into the output memory array. However, this leads to very poor performance, even when using the optimised cuBlas library for the dot product operation. The array sizes are simply not large enough for the dot product function to operate efficiently.</p>
<p>The modified algorithm can potentially perform better as it can be implemented in a single kernel that removes the need for a dot product. However, each thread will be competing to access and write to the output memory as it updates over time-steps. This will inevitably lead to race conditions accessing that data. For the parallel code to function correctly, it is necessary to employ an atomic function to read, modify, and subsequently write to the output array within each thread. CUDA provides this functionality with atomicAdd (), which ensures that only a single thread can modify the value at any given moment. The resulting kernel is presented in <xref ref-type="statement" rid="Listing_1">Listing 1</xref>. Further optimisations are given in the following sections.</p>
<p>
<statement content-type="listing" id="Listing_1">
<label>Listing 1</label>
<p>CUDA Kernel for Modifed Algorithm with atomicAdd() to avoid race conditions.<list list-type="simple">
<list-item>
<p>__<monospace>global</monospace>__ <monospace>
<bold>void</bold> updateState (<bold>float</bold>&#x2a; uNext, <bold>float</bold>&#x2a; u, float&#x2a; uPrev, <bold>float</bold>&#x2a; c1, float&#x2a; c2, <bold>float</bold>&#x2a; c3, float&#x2a; wouts, <bold>float</bold>&#x2a; input, <bold>float</bold>&#x2a; output, int b)</monospace>
</p>
</list-item>
<list-item>
<p>{</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<monospace>
<bold>int</bold> m &#x3d; blockIdx.u&#x2a;Blocksize &#x2b; threadIdx.u;</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<monospace>
<bold>for</bold> (<bold>int</bold> t &#x3d; 0; t &#x3c; buffer_size; &#x2b;&#x2b;t)</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<monospace>{</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;<monospace>uNext [m] &#x3d; u [m]&#x2a;c1 [m] &#x2b; uPrev [m]&#x2a;c2 [m] &#x2b; c3 [m]&#x2a;input [t];</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;<monospace>
<bold>float</bold> outpart &#x3d; uNext [m]&#x2a;wouts [m];</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;<monospace>
<bold>int</bold> index &#x3d; b&#x2a;buffer_size &#x2b; t;</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;<monospace>atomicAdd (&#x26;(output [index]), outpart);</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;<monospace>uPrev [m] &#x3d; u [m];</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;<monospace>u [m] &#x3d; uNext [m];</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#xa0;&#x2003;<monospace>}</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;<monospace>}</monospace>
</p>
</list-item>
</list>
</p>
</statement>
</p>
</sec>
<sec id="s4-2-3">
<title>4.2.3 Test setups for optimised GPU strategies</title>
<p>In subsequent sections, we compare per-platform optimisations applied to synthesis kernels implemented in CUDA and Metal. The systems under test are documented in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Test systems for GPU experiments.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">System</th>
<th align="center">Processor</th>
<th align="center">RAM</th>
<th align="center">GPU</th>
<th align="center">VRAM</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Windows PC</td>
<td align="center">Intel i7-12700</td>
<td align="center">32&#xa0;GB DDR4</td>
<td align="center">GeForce RTX 4070</td>
<td align="center">12&#xa0;GB GDDR5</td>
</tr>
<tr>
<td align="left">MacOS</td>
<td align="center">M2 Pro 10-core</td>
<td align="center">16&#xa0;GB Shared</td>
<td align="center">Integrated GPU, 16-core</td>
<td align="center">16&#xa0;GB Shared</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4-2-4">
<title>4.2.4 Banked memory writes plus tree-sum stage</title>
<p>The computation continues with each thread tasked to compute one mode and to write a 256- or 512-sample audio buffer. These writes are organised such that 32 threads within a warp<xref ref-type="fn" rid="fn14">
<sup>14</sup>
</xref> compute the same time sample <inline-formula id="inf65">
<mml:math id="m77">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and write in unison to adjacent memory locations. Careful consideration of memory access patterns helps to avoid bank conflicts and contention, thereby enabling higher memory throughput. This process yields numberOfModes channels of audio, with one mode per channel, that must be summed to produce a monaural output. Summation is performed within each warp and then across warps to achieve the final monaural signal. Within a warp, an efficient primitive, such as __shfl_down_sync (), may be used to conduct a tree-sum reduction, collapsing 32 channels to one in five pairwise steps. This operation results in one partially summed channel per thread group, which can be finalised on either the GPU or CPU. A demonstration of the two-stage tree-sum approach is presented below in <xref ref-type="statement" rid="Listing_2">Listing 2</xref>, as an extension of <xref ref-type="statement" rid="Listing_1">Listing 1</xref>.</p>
<p>
<statement content-type="listing" id="Listing_2">
<label>Listing 2</label>
<p>Modified kernel with specialisation of summation within each warp as &#x201c;submixes.&#x201d;<list list-type="simple">
<list-item>
<p>&#xa0;<monospace>//For simplicity, assume number of filters is a multiple of 32.</monospace>
</p>
</list-item>
<list-item>
<p>&#xa0;__<monospace>global__ <bold>void</bold> updateState (<bold>float</bold>&#x2a; uNext, <bold>float</bold>&#x2a; u, float&#x2a; uPrev, <bold>float</bold>&#x2a; c1,</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<monospace>
<bold>float</bold>&#x2a; c2, <bold>float</bold>&#x2a; c3, <bold>float</bold>&#x2a; wouts, <bold>float</bold>&#x2a; input, <bold>float</bold>&#x2a; output, int b)</monospace>
</p>
</list-item>
<list-item>
<p>
<monospace>{</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<monospace>
<bold>int</bold> m &#x3d; blockIdx.x &#x2a; blockDim.x &#x2b; threadIdx.x;</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<monospace>
<bold>int</bold> warpId &#x3d; threadIdx.x/32;</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;// <monospace>Threads within a warp each compute a filter independently.</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;// <monospace>Arrange data so memory writes are aligned.</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<monospace>for (int t &#x3d; 0; t &#x3c; BUFFER_SIZE; &#x2b;&#x2b;t)</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<monospace>{</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;<monospace>uNext [m] &#x3d; u [m]&#x2a;c1 [m] &#x2b; uPrev [m]&#x2a;c2 [m] &#x2b; c3 [m]&#x2a;input [t];</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;<monospace>
<bold>float</bold> outpart &#x3d; uNext [m]&#x2a;wouts [m];</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;<monospace>
<bold>int</bold> index &#x3d; t&#x2a;THREADGROUP_SIZE &#x2b; m;</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;<monospace>output [index] &#x2b; &#x3d; outpart;</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;<monospace>uPrev [m] &#x3d; u [m];</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;<monospace>u [m] &#x3d; uNext [m];</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<monospace>}</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;// <monospace>Sum the audio channels within the warp to mono.</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<monospace>
<bold>for</bold> (<bold>int</bold> t &#x3d; 0; t &#x3c; BUFFER_SIZE; &#x2b;&#x2b;t) {</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;<monospace>
<bold>float</bold> thread_value &#x3d; output [t&#x2a;THREADGROUP_SIZE &#x2b; m];</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;<monospace>
<bold>for</bold> (<bold>int</bold> offset &#x3d; 16; offset &#x3e;0; offset/ &#x3d; 2) {</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;<monospace>thread_value &#x2b; &#x3d; __shfl_down_sync (0xffffffff, thread_value, offset);</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;}</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;// <monospace>First thread in warp writes result.</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;<monospace>
<bold>if</bold> (m</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;<monospace>output [t] &#x3d; thread_value;</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;<monospace>}</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<monospace>}</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;// <monospace>We now transfer only the first audio channel from the device.</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;// <monospace>Filter state &#x7c;uPrev&#x7c; for each filter must be sent to the host</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;// <monospace>or persisted to global memory.</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;<monospace>}</monospace>
</p>
</list-item>
</list>
</p>
<p>This multi-stage summation approach has minimal effect when handling a small number of modes but increases the maximum synthesisable mode count by over 2.5 times compared to a single-threaded CPU summation stage. This increase is attributable to both parallelisation and reduced I/O costs. With this approach, data transfer overhead (host-to-device transfer of parameters plus device-to-host transfer of the output signals) is 12% of the total wall clock time and second-stage summation of the per-warp signals on the CPU 17%. Further optimisations of the summation stage include summing across warps on the GPU or employing multi-threading and vectorisation on the CPU side.</p>
</statement>
</p>
</sec>
<sec id="s4-2-5">
<title>4.2.5 Hybrid CPU/GPU approach</title>
<p>Metrics presented in this section synthesise all filters on the GPU, however we note that further performance gains might be obtained by using the CPU and GPU in parallel.</p>
<p>During the synthesis of modes on a GPU, the CPU remains idle. Depending on the GPU hardware, execution may encounter a bottleneck where filter updates saturate the FPU units with multiplications. If a single buffer of latency is acceptable in a real-time application, it may be possible to leverage the idle hardware by pipelining the mode computation and summation operations, thus enabling additional parallelism. For instance, audio may be synthesised as a sequence of buffer computations <inline-formula id="inf66">
<mml:math id="m78">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2026;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. While synthesising modes for buffer <inline-formula id="inf67">
<mml:math id="m79">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, post-processing, including summation to mono, may be performed for buffer <inline-formula id="inf68">
<mml:math id="m80">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. This approach allows the execution time of the faster computational task to be effectively &#x201c;hidden.&#x201d;</p>
</sec>
<sec id="s4-2-6">
<title>4.2.6 Caching</title>
<p>On Nvidia discrete GPUs, the 32 threads&#x2019; buffers of 256 or 512 samples may fit in fast shared memory local to a thread block. Our implementation explicitly requests to store data in this memory and is written to avoid bank conflicts. GPU kernels that must use larger but slower global memory may automatically benefit from caches, however we confirmed maximum throughput was achieved by keeping our working set in shared memory.</p>
<p>On the CUDA system, these optimisations enable the synthesis of a signal comprising over 600,000 modes at 48&#xa0;kHz in real time, meeting the feasibility threshold of a real-time factor below 1.0 for synthesising one second of audio. Results for this platform in isolation are presented at the end of this section in <xref ref-type="table" rid="T3">Table 3</xref>.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Number of modes computed in 0.8&#xa0;s of run-time on each GPU system over 48000 time-steps, processed in sets of <inline-formula id="inf69">
<mml:math id="m81">
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>buffer<inline-formula id="inf70">
<mml:math id="m82">
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> samples.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">System</th>
<th align="center">Buffer</th>
<th align="center">Max modes</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">RTX 4070</td>
<td align="center">256</td>
<td align="center">494k</td>
</tr>
<tr>
<td align="left">RTX 4070</td>
<td align="center">512</td>
<td align="center">618k</td>
</tr>
<tr>
<td align="left">M2 GPU</td>
<td align="center">256</td>
<td align="center">124k</td>
</tr>
<tr>
<td align="left">M2 GPU</td>
<td align="center">512</td>
<td align="center">108k</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4-2-7">
<title>4.2.7 Discrete GPU latency analysis</title>
<p>It is observed that there is significantly greater variance in latency when synthesising modes on the GPU compared to the CPU, while the FPGA system provides even more stable latency metrics. We analyze this practical consideration for the GPU platform. An experiment was conducted in which audio was processed in a series of 256- or 512-sample buffers, simulating a real-time system typical of audio production. <xref ref-type="table" rid="T4">Table 4</xref> presents the median, 95th percentile, and maximum computational latency observed across each buffer processed on the GPU system. If any of these values, along with overhead, exceeds the host audio driver&#x2019;s deadline for providing a result, a buffer underrun will occur, resulting in audio dropout in a real-time system. Latency compensation or pipelining may mitigate such issues in practical applications.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Latency statistics for processing 256-sample buffers, using CUDA system, for increasing number of modes, in milliseconds.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Modes</th>
<th align="center">Min</th>
<th align="center">Max</th>
<th align="center">Avg</th>
<th align="center">p50</th>
<th align="center">p95</th>
<th align="center">p99</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">48k</td>
<td align="center">0.17</td>
<td align="center">0.62</td>
<td align="center">0.22</td>
<td align="center">0.21</td>
<td align="center">0.31</td>
<td align="center">0.62</td>
</tr>
<tr>
<td align="left">96k</td>
<td align="center">0.24</td>
<td align="center">0.86</td>
<td align="center">0.29</td>
<td align="center">0.26</td>
<td align="center">0.43</td>
<td align="center">0.86</td>
</tr>
<tr>
<td align="left">192k</td>
<td align="center">0.48</td>
<td align="center">1.53</td>
<td align="center">0.56</td>
<td align="center">0.52</td>
<td align="center">0.76</td>
<td align="center">1.53</td>
</tr>
<tr>
<td align="left">384k</td>
<td align="center">0.72</td>
<td align="center">3.00</td>
<td align="center">0.96</td>
<td align="center">0.84</td>
<td align="center">1.84</td>
<td align="center">3.00</td>
</tr>
<tr>
<td align="left">768k</td>
<td align="center">1.83</td>
<td align="center">3.46</td>
<td align="center">2.21</td>
<td align="center">2.13</td>
<td align="center">2.78</td>
<td align="center">3.46</td>
</tr>
<tr>
<td align="left">1536k (not feasible)</td>
<td align="center">4.29</td>
<td align="center">7.50</td>
<td align="center">4.98</td>
<td align="center">4.89</td>
<td align="center">5.70</td>
<td align="center">7.50</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4-2-8">
<title>4.2.8 Shared memory system</title>
<p>The aforementioned GPU results were measured on PCs with CUDA-compatible GPUs and APIs. Modern Apple platforms contain an on-die GPU and in-package memory, which the CPU and GPU each write to directly. There may be some locking and synchronisation overhead, but a transfer from one memory to another is avoided. As an experiment, the simulation was run on an Apple M2 system, using Metal GPU compute shader code to perform the mode computation and subsequent sum to one channel. Results are displayed in <xref ref-type="table" rid="T5">Table 5</xref>. To emphasize, reported statistics for CUDA platforms include explicit host/device data transfers; there is no corresponding explicit data transfer step in the Metal kernel, but both platforms&#x2019; performance statistics include any overhead for relevant API calls and synchronisation. Due to the GPU platforms&#x2019; variance in latency between calls, we present buffer processing times here. Maximum system throughput is presented in the immediately following section, for comparison with the CPU and FPGA platforms.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Median and tail latency processing times for buffers of 256 samples, shared memory architecture.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Modes</th>
<th align="center">p50</th>
<th align="center">p95</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">12.5k</td>
<td align="center">1.38</td>
<td align="center">2.45</td>
</tr>
<tr>
<td align="left">24k</td>
<td align="center">1.49</td>
<td align="center">2.88</td>
</tr>
<tr>
<td align="left">48k</td>
<td align="center">1.93</td>
<td align="center">3.50</td>
</tr>
<tr>
<td align="left">96k</td>
<td align="center">3.61</td>
<td align="center">5.23</td>
</tr>
<tr>
<td align="left">192k</td>
<td align="center">3.78</td>
<td align="center">8.21</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4-2-9">
<title>4.2.9 Results and observations</title>
<p>With optimisations, the maximum number of modes that may be synthesised for a 1-s signal of audio in under 800 milliseconds of compute time was again determined, reserving 200 milliseconds for overhead in the calling code, operating system, or GPU driver. Times reported include data transfer to and from the GPU kernel and all required API calls. Data was processed in buffers sized similarly to those requested by real-world digital audio workstation hosts; this also allowed efficient use of GPU memory caches and thread-local memory on the CUDA platform.</p>
</sec>
</sec>
<sec id="s4-3">
<title>4.3 Field-programmable gate array</title>
<p>FPGAs are well-suited for implementing modal processors due to their parallelisation capabilities. In this context, High-Level Synthesis (HLS), specifically the Syfala toolchain (see &#xa7;2.2), was selected to conduct experiments on AMD-Xilinx FPGAs. The choice of these tools permitted the use of a similar C&#x2b;&#x2b; code input as employed in the CPU/GPU experiments while also enabling the utilisation of C-Simulation (CSIM) to verify that the output results matched those of the C&#x2b;&#x2b; reference code.</p>
<p>Additionally, the Syfala toolchain allowed us to quickly and fully implement the algorithm on different FPGA development boards, and test its output in a reliable real-time context of execution. In that regard, having a hand-tuned HDL implementation for this specific algorithm would have probably allowed us to reach better overall performances, but would have certainly been much more challenging and time-consuming to implement and stabilize.</p>
<sec id="s4-3-1">
<title>4.3.1 Metrics</title>
<p>In the context of an FPGA implementation, latency is measured by taking the critical path of the implemented DSP &#x201c;circuit,&#x201d; which will be the number of clock cycles that its longest branch takes to get a value from input to output. Another crucial metric comes into play when implementing a specific algorithm: its <italic>size</italic>, i.e., the area it occupies on the Programmable Logic (PL). The first requirement is that the algorithm&#x2019;s usage of logic resources stays below the number of available logic cells (or slices) that the FPGA chip physically possesses. Those cells are usually divided into different categories.<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf71">
<mml:math id="m83">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Digital Signal Processing (DSP) slices, mainly used for multiply-accumulate operations;</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf72">
<mml:math id="m84">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Flip Flops (FF), which are individual single clock-driven logic registers;</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf73">
<mml:math id="m85">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Look-up Tables (LUT) which are used for logic operations, multiplexers, etc.;</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf74">
<mml:math id="m86">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Block RAM (BRAM), which can be used for storing static arrays, implementing FIFOs, etc.;</p>
</list-item>
</list>If the estimation or implementation reports generated by Vitis HLS indicate that any of those cell categories is over-used, the synthesis/implementation process will eventually fail to produce an FPGA bitstream.</p>
</sec>
<sec id="s4-3-2">
<title>4.3.2 Implementation details</title>
<p>All experiments have been conducted with AMD-Xilinx Vitis HLS 2024.1 and tested on incrementally-sized AMD-Xilinx FPGA chips as listed in <xref ref-type="table" rid="T6">Table 6</xref>, in order to be able to benefit from a higher number of logic resources each time a maximum of modes and/or logic resources was reached for a specific chip.</p>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>Hardware resources in a number of slices of the different FPGAs used for our experiments.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">FPGA chip</th>
<th align="center">DSP</th>
<th align="center">FF</th>
<th align="center">LUT</th>
<th align="center">BRAM</th>
<th align="center">URAM</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">xc7z010clg400</td>
<td align="center">80</td>
<td align="center">35200</td>
<td align="center">17600</td>
<td align="center">120</td>
<td align="center">0</td>
</tr>
<tr>
<td align="center">xc7z020clg400</td>
<td align="center">220</td>
<td align="center">106400</td>
<td align="center">53200</td>
<td align="center">280</td>
<td align="center">0</td>
</tr>
<tr>
<td align="center">xc7z035ffg676</td>
<td align="center">900</td>
<td align="center">343800</td>
<td align="center">171900</td>
<td align="center">1,000</td>
<td align="center">0</td>
</tr>
<tr>
<td align="center">xc7z100ffg900</td>
<td align="center">2020</td>
<td align="center">554800</td>
<td align="center">277400</td>
<td align="center">1,510</td>
<td align="center">0</td>
</tr>
<tr>
<td align="center">xczu3eg-sfvc784</td>
<td align="center">320</td>
<td align="center">141120</td>
<td align="center">70560</td>
<td align="center">432</td>
<td align="center">0</td>
</tr>
<tr>
<td align="center">xczu15eg-ffvb1156</td>
<td align="center">3,528</td>
<td align="center">682560</td>
<td align="center">341280</td>
<td align="center">1,488</td>
<td align="center">112</td>
</tr>
<tr>
<td align="center">xczu19eg-ffvc1760</td>
<td align="center">1968</td>
<td align="center">1045440</td>
<td align="center">522720</td>
<td align="center">1968</td>
<td align="center">128</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>A full implementation and validation process (including sound testing) was conducted for FPGA development boards in our possession, including the <italic>xc7z010clg400-1</italic>, <italic>xc7z020clg400-1</italic>, and <italic>xczu3eg-sfvc784-1-e</italic> FPGAs. For the others, only &#x201c;estimate reports&#x201d; were available, which do not always guarantee that the implementation will eventually fit on the FPGA, especially if resource utilisation is getting close to saturation. Consequently, in order to obtain a safer margin, experiments reporting a resource utilisation of more than 80% on one or several logic cell categories were discarded for those specific chips.</p>
<p>Regarding latency, all experiments were configured to run with a 48&#xa0;kHz sample rate and a 122.88&#xa0;MHz FPGA clock rate, which was chosen in order to match the traditional 256<italic>fs</italic> or 384<italic>fs</italic> based rates used for operating with audio codecs. Higher clock-rates, such as 184.32&#xa0;MHz or 245.76&#xa0;MHz were also experimented, but resulted in errors and <italic>timing violations</italic> throughout the HLS process, which we were not able to fully resolve yet. Consequently, the maximum latency for the processing of a single sample was in this context of 2,560 clock cycles. Buffer sizes from 24 to 1,024 samples were used in order to keep the latency per sample below maximum.</p>
<p>
<statement content-type="listing" id="Listing_3">
<label>Listing 3</label>
<p>Vitis HLS implementation of modified algorithm.<list list-type="simple">
<list-item>
<p>&#x2003;<monospace>
<bold>static float</bold> u [modesNumber];</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;<monospace>
<bold>static float</bold> uPrev [modesNumber];</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;<monospace>
<bold>static float</bold> uNext [modesNumber];</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;<monospace>
<bold>int</bold> c &#x3d; 0;</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;<monospace>
<bold>for</bold> (<bold>int</bold> m &#x3d; 0; m &#x3c; modesNumber; m&#x2b;&#x2b;) {</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<monospace>&#x23;pragma HLS pipeline</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<monospace>float c1 &#x3d; coeffs [c&#x2b;&#x2b;];</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<monospace>float c2 &#x3d; coeffs [c&#x2b;&#x2b;];</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<monospace>float c3 &#x3d; coeffs [c&#x2b;&#x2b;];</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<monospace>float modes_out &#x3d; coeffs [c&#x2b;&#x2b;];</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<monospace>for (int n &#x3d; 0; n &#x3c; BUFFER_NSAMPLES; &#x2b;&#x2b;n) {</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;<monospace>&#x23;pragma HLS unroll</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;<monospace>uNext [m] &#x3d; c1 &#x2a;u [m]</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;<monospace>&#x2b; c2 &#x2a; uPrev [m]</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;<monospace>&#x2b; c3 &#x2a; input [n];</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;<monospace>uPrev [m] &#x3d; u [m];</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;<monospace>u [m] &#x3d; uNext [m];</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;<monospace>output [n] &#x2b; &#x3d; uNext [m] &#x2a; modes_out;</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<monospace>}</monospace>
</p>
</list-item>
<list-item>
<p>&#x2003;<monospace>}</monospace>
</p>
</list-item>
</list>
</p>
<p>Initialisation of coefficient arrays c1, c2, c3 and modes_out was done on the ARM CPU (which is on the same System on Chip as the FPGA), stored in DDR memory and shared with the PL through the ARM <italic>Advanced Microcontroller Bus Architecture</italic> (AMBA), using the <italic>Advanced eXtensible Interface</italic> (AXI4) protocol. This allowed to prevent logic resource saturation when the number of modes became too important for all coefficients to be directly stored and read in the PL Block-RAMs or LUT-RAMs. Furthermore, it also allowed freeing DSP slices for expensive operations (such as cosine, exponential and square-root functions) that would have been used only once but permanently implemented on the PL. These coefficients were stored in an interleaved way, so they could be retrieved from the DSP kernel with a single burst request occurring for each mode iteration, somewhat limiting the impact on latency. There were no other read/write accesses to DDR memory. The intermediate storage arrays (u [N], uPrev [N], and uNext [N]) were, on the other hand, directly allocated in the programmable logic (PL), in either Block-RAMs or LUT-RAMs. Finally, as shown in <xref ref-type="statement" rid="Listing_3">Listing 3</xref>, DSP computation loops were &#x201c;inverted,&#x201d; resulting in a positive performance impact by reducing interdependencies between logic cells and enabling the parallel processing of a buffer of samples rather than an array of coefficients. Such an approach would have been challenging to implement on smaller FPGA chips due to limited resources. To enforce this strategy, the <italic>pipeline</italic> and <italic>unroll</italic> HLS pragmas were applied to each loop.</p>
</statement>
</p>
</sec>
<sec id="s4-3-3">
<title>4.3.3 Code verification</title>
<p>In order to properly ensure that the DSP kernel is valid and produces - from the same inputs - the same outputs as the original C&#x2b;&#x2b; reference code, Vitis HLS&#x2019; C-Simulation (CSIM) feature was used. Vitis HLS maintains the order of operations performed in the C code when synthesizing float and double types to ensure that the results are the same as the C simulation, unless optimisations are explicitly requested. To quantify numerical accuracy, we compared the FPGA-generated outputs on a <italic>Digilent Zybo Z7-20</italic> with the C&#x2b;&#x2b; reference implementation across 48,000 samples. The results showed a mean absolute error of 1.85 <inline-formula id="inf75">
<mml:math id="m87">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mspace width="0.3333em"/>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>6</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, a maximum absolute error of 9.4 <inline-formula id="inf76">
<mml:math id="m88">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mspace width="0.3333em"/>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>6</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, and a signal-to-noise ratio (SNR) of <italic>87.25&#xa0;dB</italic>. These minor discrepancies stem from floating-point implementation variations, which are not likely to have a perceptible impact on the resulting audio.</p>
</sec>
<sec id="s4-3-4">
<title>4.3.4 Results and observations</title>
<p>In this configuration, the addition of a single <italic>mode</italic> increased latency by approximately 10 clock cycles. Increasing the buffer size provided a means to offset this latency penalty by enabling parallelisation; however, this also led to a substantial rise in the utilisation of logic resources, as these were duplicated for the processing of each sample. The final results of our experiments and simulations, in terms of maximum number of modes for each targeted FPGA chip, are detailed in <xref ref-type="table" rid="T7">Table 7</xref>.</p>
<table-wrap id="T7" position="float">
<label>TABLE 7</label>
<caption>
<p>Maximum number of synthesised modes in real-time on different FPGAs and corresponding resource usage. Latency corresponds here to the time it takes to generate a sample to meet real-time conditions. If latency exceeds 100%, then the program can not be executed in real-time.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">FPGA</th>
<th align="center">Modes</th>
<th align="center">Buffer</th>
<th align="center">Latency</th>
<th align="center">DSP</th>
<th align="center">FF</th>
<th align="center">LUT</th>
<th align="center">BRAM</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">xc7z010clg400</td>
<td align="center">12,000</td>
<td align="center">24</td>
<td align="center">98%</td>
<td align="center">68%</td>
<td align="center">51%</td>
<td align="center">62%</td>
<td align="center">67%</td>
</tr>
<tr>
<td align="center">xc7z020clg400</td>
<td align="center">30,000</td>
<td align="center">64</td>
<td align="center">92%</td>
<td align="center">65%</td>
<td align="center">40%</td>
<td align="center">50%</td>
<td align="center">52%</td>
</tr>
<tr>
<td align="center">xczu3eg-sfvc784</td>
<td align="center">45,000</td>
<td align="center">80</td>
<td align="center">88%</td>
<td align="center">88%</td>
<td align="center">44%</td>
<td align="center">70%</td>
<td align="center">45%</td>
</tr>
<tr>
<td align="center">xc7z035ffg676</td>
<td align="center">110,000</td>
<td align="center">310</td>
<td align="center">97%</td>
<td align="center">54%</td>
<td align="center">34%</td>
<td align="center">80%</td>
<td align="center">54%</td>
</tr>
<tr>
<td align="center">xc7z100ffg900</td>
<td align="center">180,000</td>
<td align="center">500</td>
<td align="center">99%</td>
<td align="center">39%</td>
<td align="center">33%</td>
<td align="center">79%</td>
<td align="center">69%</td>
</tr>
<tr>
<td align="center">xczu15eg-ffvb1156</td>
<td align="center">300,000</td>
<td align="center">800</td>
<td align="center">88%</td>
<td align="center">41%</td>
<td align="center">43%</td>
<td align="center">77%</td>
<td align="center">73%</td>
</tr>
<tr>
<td align="center">xczu19eg-ffvc1760</td>
<td align="center">350,000</td>
<td align="center">700</td>
<td align="center">98%</td>
<td align="center">78%</td>
<td align="center">30%</td>
<td align="center">49%</td>
<td align="center">66%</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
</sec>
<sec id="s5">
<title>5 Discussion and future directions</title>
<p>CPUs, GPUs, and FPGAs function in a very different way and potentially in very different contexts. In that regard, it is not necessarily straightforward to provide an objective comparison between them.</p>
<p>An important consideration here is that this study has focused on achieving real-time performance for the largest modal model across various processors without accounting for the potential parallel operation of other system elements. In other words, the utility of running such computationally intensive algorithms may be limited if other processes&#x2014;such as a Digital Audio Workstation (DAW), graphical interface, or physical user interface&#x2014;are not running concurrently. While this limitation may be less pronounced when using GPUs or FPGAs, typically employed as &#x201c;hardware accelerators&#x201d; alongside a CPU, preserving capacity on a standalone CPU for additional tasks is critical, particularly as an Operating System (OS) is likely to be running.</p>
<p>One of the main observations that can be made from the results presented in the previous sections is that recent CPUs do provide unexpectedly good performances in the context of modal synthesis, potentially competing with processors providing a higher level of parallelisation, such as FPGAs and GPUs. That being said, and despite the aforementioned limitations in terms of data transfer between the GPU and its associated CPU, GPUs do provide the best performances compared to CPUs and FPGAs, allowing for more than 600k modes to be synthesised in real-time and leaving the associated CPU free to carry out other tasks. It is also worth noting that the multi-threaded CPU and GPU approaches may each realize tens of thousands of modes in a context where the use of the resources is not mutually exclusive. In that regard, future work might explore a hybrid system that divides work among the two simultaneously.</p>
<p>FPGAs appear to yield &#x201c;less impressive&#x201d; results within the context of modal synthesis, compounded by the high potential cost of larger FPGA models, such as the <italic>xczu19eg-ffvc1760</italic>, which are likely considerably more expensive than the CPUs and GPUs evaluated in this study. However, one metric not addressed here, where FPGAs demonstrate superiority, is audio latency. Despite the focus on computational efficiency in the configurations presented in &#xa7;4.3.4, it is feasible to eliminate buffering on an FPGA, thereby achieving outstanding latency performance compared to other processors.</p>
<p>An additional consideration is that the current study employs HLS for FPGA programming to provide a more balanced comparison between FPGAs, GPUs, and CPUs. Rewriting the modal synthesis algorithm entirely in a hardware description language, using fixed-point rather than floating-point arithmetic may yield improved performance over the results presented here. A compromise approach could involve the use of optimised floating-point hardware operators, such as those provided by FloPoCo (<xref ref-type="bibr" rid="B15">De Dinechin and Pasca, 2011</xref>).</p>
</sec>
<sec sec-type="conclusion" id="s6">
<title>6 Conclusion</title>
<p>This study has evaluated the performance of three principal processor types&#x2014;CPUs, GPUs, and FPGAs&#x2014;in the context of modal synthesis and processing. For smaller-scale models, CPUs emerge as the most practical platform owing to their widespread availability, adaptability, and flexibility, particularly in recent models with enhanced computational capabilities. When greater computational power is required, GPUs can serve as effective accelerators alongside CPUs, providing a scalable solution for larger model processing in many personal computers, where high-performance GPUs are now commonly installed. Although FPGAs are less accessible and typically not integrated into standard PCs, they offer unmatched audio latency performance, which may be advantageous in specific real-time audio processing applications.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>RM: Writing&#x2013;original draft, Writing&#x2013;review and editing. MD: Writing&#x2013;original draft, Writing&#x2013;review and editing. PC: Writing&#x2013;original draft, Writing&#x2013;review and editing. TS: Writing&#x2013;original draft, Writing&#x2013;review and editing. CW: Writing&#x2013;original draft, Writing&#x2013;review and editing. RR: Writing&#x2013;original draft, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work received funding from the European Research Council (ERC) under the European Union&#x2019;s Horizon 2020 Research and Innovation programme Grant agreement No. 950084 NEMUS.</p>
</sec>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s11">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn id="fn1">
<label>1</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://gpu.audio/">https://gpu.audio/</ext-link>
</p>
</fn>
<fn id="fn2">
<label>2</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://www.acustica-audio.com/">https://www.acustica-audio.com/</ext-link>
</p>
</fn>
<fn id="fn3">
<label>3</label>
<p>SoC: System on Chip.</p>
</fn>
<fn id="fn4">
<label>4</label>
<p>The term &#x201c;hardware programming&#x201d; is commonly used in the context of FPGA programming.</p>
</fn>
<fn id="fn5">
<label>5</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://novationmusic.com/en/synths/summit">https://novationmusic.com/en/synths/summit</ext-link>
</p>
</fn>
<fn id="fn6">
<label>6</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://en.antelopeaudio.com/">https://en.antelopeaudio.com/</ext-link>
</p>
</fn>
<fn id="fn7">
<label>7</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://www.udo-audio.com/\#introduction">https://www.udo-audio.com/\&#x23;introduction</ext-link>
</p>
</fn>
<fn id="fn8">
<label>8</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://www.futur3soundz.com/">https://www.futur3soundz.com/</ext-link>
</p>
</fn>
<fn id="fn9">
<label>9</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://www.audinate.com/">https://www.audinate.com/</ext-link>
</p>
</fn>
<fn id="fn10">
<label>10</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://www.mathworks.com/products/simulink.html">https://www.mathworks.com/products/simulink.html</ext-link>
</p>
</fn>
<fn id="fn11">
<label>11</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://www.mathworks.com/products/hdl-coder.html">https://www.mathworks.com/products/hdl-coder.html</ext-link>
</p>
</fn>
<fn id="fn12">
<label>12</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://inria-emeraude.github.io/syfala/">https://inria-emeraude.github.io/syfala/</ext-link>
</p>
</fn>
<fn id="fn13">
<label>13</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://inria-emeraude.github.io/syfala/tutorials/cpp-tutorial-advanced/">https://inria-emeraude.github.io/syfala/tutorials/cpp-tutorial-advanced/</ext-link>
</p>
</fn>
<fn id="fn14">
<label>14</label>
<p>A CUDA warp is a collection of threads concurrently executing the same code on different pieces of data.</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abel</surname>
<given-names>J. S.</given-names>
</name>
<name>
<surname>Coffin</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Spratt</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Human&#x2013;made rock mixes feature tight relations between spectrum and loudness (JAES volume 62 issue 10 pp. 643-653; october 2014)</article-title>. <source>J. Audio Eng. Soc.</source> <volume>62</volume>, <fpage>643</fpage>&#x2013;<lpage>653</lpage>. <pub-id pub-id-type="doi">10.17743/jaes.2014.0039</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Adrien</surname>
<given-names>J.-M.</given-names>
</name>
</person-group> (<year>1991</year>). &#x201c;<article-title>The missing link: modal synthesis</article-title>,&#x201d; in <source>Representations of musical signals</source> (<publisher-loc>Cambridge, MA, USA</publisher-loc>: <publisher-name>MIT Press</publisher-name>), <fpage>269</fpage>&#x2013;<lpage>298</lpage>.</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arcas</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Chaigne</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>On the quality of plate reverberation</article-title>. <source>Appl. Acoust.</source> <volume>71</volume>, <fpage>147</fpage>&#x2013;<lpage>156</lpage>. <pub-id pub-id-type="doi">10.1016/j.apacoust.2009.07.013</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Avitabile</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). <source>Modal testing: a practitioner&#x2019;s guide</source>. <publisher-loc>Hoboken, New Jersey</publisher-loc>: <publisher-name>John Wiley &#x26; Sons</publisher-name>.</citation>
</ref>
<ref id="B5">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Bakhoda</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>G. L.</given-names>
</name>
<name>
<surname>Fung</surname>
<given-names>W. W.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Aamodt</surname>
<given-names>T. M.</given-names>
</name>
</person-group> (<year>2009</year>). &#x201c;<article-title>Analyzing cuda workloads using a detailed gpu simulator</article-title>,&#x201d; in <source>
<italic>2009 IEEE international symposium on performance analysis of systems and software</italic> (IEEE)</source>, <fpage>163</fpage>&#x2013;<lpage>174</lpage>.</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bank</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ramos</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Improved pole positioning for parallel filters based on spectral smoothing and multiband warping</article-title>. <source>IEEE Signal Process. Lett.</source> <volume>18</volume>, <fpage>299</fpage>&#x2013;<lpage>302</lpage>. <pub-id pub-id-type="doi">10.1109/lsp.2011.2124456</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Bilbao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Arcas</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Chaigne</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>A physical model for plate reverberation</article-title>. In <source>2006 IEEE international conference on acoustics speech and signal processing proceedings</source>. vol. <volume>5</volume>, V&#x2013;V</citation>
</ref>
<ref id="B8">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Bilbao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Desvages</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ducceschi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hamilton</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Harrison-Harsley</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Torin</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <source>The ness project: physical modeling, algorithms and sound synthesis</source>. <publisher-name>Computer Music Journal. Cambridge, MA: MIT Press</publisher-name>.</citation>
</ref>
<ref id="B9">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Boyd</surname>
<given-names>J. P.</given-names>
</name>
</person-group> (<year>2001</year>). <source>Chebyshev and Fourier spectral methods</source>. <publisher-loc>Mineola, New York, USA</publisher-loc>: <publisher-name>Dover</publisher-name>.</citation>
</ref>
<ref id="B10">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Cannon</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Saniie</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Modular delay audio effect system on FPGA</article-title>,&#x201d; in <source>Proceedings of the 2022 IEEE international conference on electro information technology (eIT)</source> (<publisher-loc>Mankato, USA</publisher-loc>), <fpage>248</fpage>&#x2013;<lpage>251</lpage>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chhetri</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Poudel</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ghimire</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shresthamali</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sharma</surname>
<given-names>D. K.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Implementation of audio effect generator in FPGA</article-title>. <source>Nepal J. Sci. Technol.</source> <volume>15</volume>, <fpage>89</fpage>&#x2013;<lpage>98</lpage>. <pub-id pub-id-type="doi">10.3126/njst.v15i1.12022</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cie&#x15b;li&#x144;ski</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>On the exact discretization of the classical harmonic oscillator equation</article-title>. <source>J. Differ. Equations Appl.</source> <volume>17</volume>, <fpage>1673</fpage>&#x2013;<lpage>1694</lpage>. <pub-id pub-id-type="doi">10.1080/10236191003730563</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Cochard</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Popoff</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Michon</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Risset</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Programming FPGA platforms for real-time audio signal processing in C&#x2b;&#x2b;</article-title>,&#x201d; in <source>Proceedings of the 2024 sound and music computing conference (SMC-24)</source> (<publisher-loc>Porto, Portugal</publisher-loc>).</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Debut</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Antunes</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Physical synthesis of six-string guitar plucks using the Udwadia-Kalaba modal formulation</article-title>. <source>J. Acoust. Soc. Am.</source> <volume>148</volume>, <fpage>575</fpage>&#x2013;<lpage>587</lpage>. <pub-id pub-id-type="doi">10.1121/10.0001635</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>De Dinechin</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Pasca</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Designing custom arithmetic data paths with flopoco</article-title>. <source>IEEE Des. &#x26; Test Comput.</source> <volume>28</volume>, <fpage>18</fpage>&#x2013;<lpage>27</lpage>. <pub-id pub-id-type="doi">10.1109/mdt.2011.44</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Deulkar</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Kolhare</surname>
<given-names>N. R.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>FPGA implementation of audio and video processing based on Zedboard</article-title>,&#x201d; in <source>Proceedings of the 2020 international conference on smart innovations in design, environment, management, planning and computing (ICSIDEMPC)</source> (<publisher-loc>Aurangabad, India</publisher-loc>: <publisher-name>IEEE</publisher-name>, <fpage>305</fpage>&#x2013;<lpage>310</lpage>.</citation>
</ref>
<ref id="B17">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Dragoi</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Anghel</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Stanciu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Paleologu</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Efficient FPGA implementation of classic audio effects</article-title>,&#x201d; in <source>
<italic>Proceedings of the 2021 13th international Conference on electronics, Computers and artificial intelligence (ECAI)</italic> (pitesti, Romania: ieee)</source>.</citation>
</ref>
<ref id="B18">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ducceschi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bilbao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Webb</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Real-time modal synthesis of nonlinearly interconnected networks</article-title>,&#x201d; in <source>Proceedings of the 26th international conference on digital audio effects (DAFx)</source> (<publisher-loc>Copenhagen, Denmark</publisher-loc>), <fpage>53</fpage>&#x2013;<lpage>60</lpage>.</citation>
</ref>
<ref id="B19">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ducceschi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Webb</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Plate reverberation: towards the development of a real-time physical model for the working musician</article-title>,&#x201d; in <source>Procedings of the 22nd international congress on acoustics ICA 2016</source>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Eckel</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>1995</year>). <article-title>Sound synthesis by physical modelling with modalys</article-title>. <source>Proc. ISMA&#x2019;</source> <volume>95</volume>, <fpage>478</fpage>&#x2013;<lpage>482</lpage>.</citation>
</ref>
<ref id="B21">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Jadhao</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Singh Patel</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Hardware architecture of digital drum kit using FPGA</article-title>,&#x201d; in <source>
<italic>Proceedings of the 2020 IEEE international Conference on advent Trends in multidisciplinary Research and innovation (ICATMRI)</italic> (buldhana, India: ieee)</source>, <fpage>1</fpage>&#x2013;<lpage>4</lpage>.</citation>
</ref>
<ref id="B22">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kereliuk</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Herman</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Little Ferry</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Wedelich</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gillespie</surname>
<given-names>D. J.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Modal analysis of room impulse responses using subband esprit</article-title>,&#x201d; in <source>
<italic>Proceedings of the 21st international Conference on digital audio effects (DAFx)</italic> (aveiro, Portugal)</source>.</citation>
</ref>
<ref id="B23">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kirk</surname>
<given-names>D. B.</given-names>
</name>
<name>
<surname>Wen-Mei</surname>
<given-names>W. H.</given-names>
</name>
</person-group> (<year>2016</year>). <source>Programming massively parallel processors: a hands-on approach</source>. <publisher-loc>Burlington, MA</publisher-loc>: <publisher-name>Morgan kaufmann</publisher-name>.</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lahti</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sj&#xf6;vall</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Vanne</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>H&#xe4;m&#xe4;l&#xe4;inen</surname>
<given-names>T. D.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Are we there yet? a study on the state of high-level synthesis</article-title>. <source>IEEE Trans. Computer-Aided Des. Integr. Circuits Syst.</source> <volume>38</volume>, <fpage>898</fpage>&#x2013;<lpage>911</lpage>. <pub-id pub-id-type="doi">10.1109/TCAD.2018.2834439</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Maestre</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Abel</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Scavone</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Constrained pole optimization for modal reverberation</article-title>,&#x201d; in <source>Proceedings of the 20th international conference on digital audio effects (DAFx)</source> (<publisher-loc>Edinburgh, UK</publisher-loc>), <fpage>381</fpage>&#x2013;<lpage>388</lpage>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maestre</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Scavone</surname>
<given-names>G. P.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>J. O.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Design of recursive digital filters in parallel form by linearly constrained pole optimization</article-title>. <source>IEEE Signal Process. Lett.</source> <volume>23</volume>, <fpage>1547</fpage>&#x2013;<lpage>1550</lpage>. <pub-id pub-id-type="doi">10.1109/lsp.2016.2605626</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>McQuillan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>van Walstijn</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Modal spring reverb based on discretisation of the thin helical spring model</article-title>,&#x201d; in <source>Proceedings of the 24th international conference on digital audio effects (DAFx)</source> (<publisher-loc>Vienna, Austria</publisher-loc>), <fpage>191</fpage>&#x2013;<lpage>198</lpage>.</citation>
</ref>
<ref id="B28">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Meirovitch</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2010</year>). <source>Fundamentals of vibrations</source>. <publisher-loc>Long Grove, Illinois, USA</publisher-loc>: <publisher-name>Waveland Press</publisher-name>.</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Motuk</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Woods</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Bilbao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>McAllister</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Design methodology for real-time FPGA-based sound synthesis</article-title>. <source>IEEE Trans. Signal Process.</source> <volume>55</volume>, <fpage>5833</fpage>&#x2013;<lpage>5845</lpage>. <pub-id pub-id-type="doi">10.1109/tsp.2007.898785</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Orlarey</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fober</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Letz</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2009</year>). <source>Faust: an efficient functional approach to DSP programming</source>. <publisher-loc>Paris, France</publisher-loc>: <publisher-name>New Computational Paradigms for Computer Music</publisher-name>, <fpage>65</fpage>&#x2013;<lpage>96</lpage>.</citation>
</ref>
<ref id="B31">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Pierce</surname>
<given-names>A. D.</given-names>
</name>
</person-group> (<year>2019</year>). <source>Acoustics: an introduction to its physical principles and applications</source>. <publisher-loc>Cham, Switzerland</publisher-loc>: <publisher-name>Springer Nature</publisher-name>.</citation>
</ref>
<ref id="B32">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Popoff</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Michon</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Risset</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Enabling affordable and scalable audio spatialization with multichannel audio expansion boards for FPGA</article-title>,&#x201d; in <source>
<italic>Proceedings of the 2024 Sound and music computing conference</italic> (porto, Portugal)</source>.</citation>
</ref>
<ref id="B33">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Popoff</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Michon</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Risset</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Orlarey</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Letz</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Towards an FPGA-based compilation flow for ultra-low latency audio signal processing</article-title>,&#x201d; in <source>
<italic>Proceedings of the 2022 Sound and music Computing conference SMC-22</italic> saint-&#xe9;tienne, France</source>.</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rabenstein</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Trautmann</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Digital sound synthesis of string instruments with the functional transformation method</article-title>. <source>Signal Process.</source> <volume>83</volume>, <fpage>1673</fpage>&#x2013;<lpage>1688</lpage>. <pub-id pub-id-type="doi">10.1016/s0165-1684(03)00083-5</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rau</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Abel</surname>
<given-names>J. S.</given-names>
</name>
<name>
<surname>James</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Julius</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Electric-to-acoustic pickup processing for string instruments: an experimental study of the guitar with a hexaphonic pickup</article-title>. <source>J. Acoust. Soc. Am.</source> <volume>150</volume>, <fpage>385</fpage>&#x2013;<lpage>397</lpage>. <pub-id pub-id-type="doi">10.1121/10.0005540</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Renney</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Mitchell</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Gaster</surname>
<given-names>B. R.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>There and back again: the practicality of gpu accelerated digital audio</article-title>,&#x201d; in <source>Nime</source>, <fpage>202</fpage>&#x2013;<lpage>207</lpage>.</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Roy</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Kailath</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>1989</year>). <article-title>Esprit-estimation of signal parameters via rotational invariance techniques</article-title>. <source>IEEE Trans. Acoust. speech, signal Process.</source> <volume>37</volume>, <fpage>984</fpage>&#x2013;<lpage>995</lpage>. <pub-id pub-id-type="doi">10.1109/29.32276</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Russo</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ducceschi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bilbao</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Efficient simulation of the bowed string in modal form</article-title>,&#x201d; in <source>Proceedings of the 25th international conference on digital audio effects (DAFx)</source> (<publisher-loc>Vienna, Austria</publisher-loc>), <fpage>122</fpage>&#x2013;<lpage>129</lpage>.</citation>
</ref>
<ref id="B39">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Russo</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ducceschi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bilbao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hamilton</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Efficient simulation of acoustic physical models with nonlinear dissipation</article-title>,&#x201d; in <source>Proceedings of the sound and music computing conference</source> (<publisher-loc>Stockholm, Sweden</publisher-loc>), <fpage>125</fpage>&#x2013;<lpage>131</lpage>.</citation>
</ref>
<ref id="B40">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Savioja</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>V&#xe4;lim&#xe4;ki</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Smith III</surname>
<given-names>J. O.</given-names>
</name>
</person-group> (<year>2010</year>). &#x201c;<article-title>Real-time additive synthesis with one million sinusoids using a gpu</article-title>,&#x201d; in <source>Audio engineering society convention</source>, <volume>128</volume>. <publisher-loc>London</publisher-loc>: <publisher-name>Audio Engineering Society</publisher-name>.</citation>
</ref>
<ref id="B41">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Schlecht</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Parker</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sch&#xe4;fer</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Rabenstein</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Physical modeling using recurrent neural networks with fast convolutional layers</article-title>,&#x201d; in <source>Proceedings of the 25th international conference on digital audio effects (DAFx)</source> (<publisher-loc>Vienna, Austria</publisher-loc>), <fpage>138</fpage>&#x2013;<lpage>145</lpage>.</citation>
</ref>
<ref id="B42">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Skare</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Abel</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Real-time modal synthesis of crash cymbals with nonlinear approximations, using a gpu</article-title>,&#x201d; in <source>
<italic>Proceedings of the 22nd international Conference on digital audio effects (DAFx)</italic> (birmingham, UK)</source>.</citation>
</ref>
<ref id="B43">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Vaca</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Jefferies</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>An open audio processing platform with Zync FPGA</article-title>,&#x201d; in <source>
<italic>Proceedings of the 2019 IEEE international Symposium on Measurement and Control in robotics (ISMCR)</italic> (houston, Texas)</source>.</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Valimaki</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Parker</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>Savioja</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>J. O.</given-names>
</name>
<name>
<surname>Abel</surname>
<given-names>J. S.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Fifty years of artificial reverberation</article-title>. <source>IEEE Trans. Audio, Speech, Lang. Process.</source> <volume>20</volume>, <fpage>1421</fpage>&#x2013;<lpage>1448</lpage>. <pub-id pub-id-type="doi">10.1109/tasl.2012.2189567</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vannoy</surname>
<given-names>T. C.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Enabling rapid prototyping of audio signal processing systems using system-on-chip field programmable gate arrays</article-title>. <comment>Masters Thesis</comment>
</citation>
</ref>
<ref id="B46">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>van Walstijn</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Numerical calculation of modal spring reverb parameters</article-title>,&#x201d; in <source>Proceedings of the 23rd international conference on digital audio effects (DAFx)</source> (<publisher-loc>Vienna, Austria</publisher-loc>).</citation>
</ref>
<ref id="B47">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>van Walstijn</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bridges</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Mehes</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>A real-time synthesis oriented tanpura model</article-title>,&#x201d; in <source>Proceedings of the 19th international conference on digital audio effects (DAFx)</source> (<publisher-loc>Brno, Czech Republic</publisher-loc>).</citation>
</ref>
<ref id="B48">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Verstraelen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kuper</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Smit</surname>
<given-names>G. J. M.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Declaratively programmable ultra-low latency audio effects processing on FPGA</article-title>,&#x201d; in <source>
<italic>Proceedings of the 17th international Conference on digital audio effects (DAFx)</italic> (erlangen, Germany)</source>.</citation>
</ref>
<ref id="B49">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Willemsen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Serafin</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jensen</surname>
<given-names>J. R.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Virtual analog simulation and extensions of plate reverberation</article-title>,&#x201d; in <source>
<italic>Proceedings of the Sound and music computing conference</italic> (espoo, Finland)</source>, <fpage>314</fpage>&#x2013;<lpage>319</lpage>.</citation>
</ref>
</ref-list>
</back>
</article>