<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Virtual Real.</journal-id>
<journal-title>Frontiers in Virtual Reality</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Virtual Real.</abbrev-journal-title>
<issn pub-type="epub">2673-4192</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1391987</article-id>
<article-id pub-id-type="doi">10.3389/frvir.2024.1391987</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Virtual Reality</subject>
<subj-group>
<subject>Technology and Code</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Networked microcontrollers for accessible, distributed spatial audio</article-title>
<alt-title alt-title-type="left-running-head">Rushton et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/frvir.2024.1391987">10.3389/frvir.2024.1391987</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Rushton</surname>
<given-names>Thomas Albert</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2528628/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Michon</surname>
<given-names>Romain</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2701869/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Serafin</surname>
<given-names>Stefania</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/308282/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Risset</surname>
<given-names>Tanguy</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Letz</surname>
<given-names>St&#xe9;phane</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>I<institution>nria, INSA Lyon, CITI, EA3720</institution>, <addr-line>Villeurbanne</addr-line>, <country>France</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Architecture, Design and Media Technology</institution>, <institution>Aalborg University</institution>, <addr-line>Copenhagen</addr-line>, <country>Denmark</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>GRAME-CNCM, INSA Lyon, Inria, CITI, EA3720</institution>, <addr-line>Villeurbanne</addr-line>, <country>France</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1753102/overview">Alessandro Pozzebon</ext-link>, University of Padua, Italy</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2721677/overview">Sven Ubik</ext-link>, Czech Education and Scientific Net work, Czechia</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2834023/overview">Jozef Pap&#xe1;n</ext-link>, University of &#x17d;ilina, Slovakia</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Thomas Albert Rushton, <email>thomas.rushton@inria.fr</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>08</day>
<month>11</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>5</volume>
<elocation-id>1391987</elocation-id>
<history>
<date date-type="received">
<day>26</day>
<month>02</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>07</day>
<month>10</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Rushton, Michon, Serafin, Risset and Letz.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Rushton, Michon, Serafin, Risset and Letz</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>State-of-the-art systems for spatial and immersive audio are typically very costly, being reliant on specialist audio hardware capable of performing computationally intensive signal processing and delivering output to many tens, if not hundreds, of loudspeakers. Centralised systems of this sort suffer from limited accessibility due to their inflexibility and expense. Building on the research of the past few decades in the transmission of audio data over computer networks, and the emergence in recent years of increasingly capable, low-cost microcontroller-based development platforms with support for both networking and audio functionality, we present a prototype decentralised, modular alternative. Having previously explored the feasibility of running a microcontroller device as a networked audio client, here we describe the development of a client-server system with improved scalability via multicast data transmission. The system operates on ubiquitous, commonplace computing and networking equipment, with a view to it being a simple, versatile, and highly-accessible platform, capable of granting users the freedom to explore audio spatialisation approaches at vastly reduced expense. Though faced by significant technical challenges, particularly with regard to maintaining synchronicity between distributed audio processors, the system produces perceptually plausible results. Findings are commensurate with a capability, with further development and research, to disrupt and democratise the fields of spatial and immersive audio.</p>
</abstract>
<kwd-group>
<kwd>spatial audio</kwd>
<kwd>networked audio</kwd>
<kwd>distributed systems</kwd>
<kwd>wave field synthesis</kwd>
<kwd>microcontroller</kwd>
<kwd>accessibility</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Technologies for VR</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Recent developments in virtual and augmented reality technologies and object-based audio have led to an acceleration in interest in the synthesis of virtual sound fields via approaches such as Wave Field Synthesis (WFS) and Higher Order Ambisonics (HOA) (<xref ref-type="bibr" rid="B11">Berkhout et al., 1993</xref>; <xref ref-type="bibr" rid="B4">Ahrens et al., 2008</xref>; <xref ref-type="bibr" rid="B23">Daniel et al., 2003</xref>; <xref ref-type="bibr" rid="B31">Frank et al., 2015</xref>). These techniques call for the deployment of large numbers of loudspeakers, and <italic>in situ</italic> installations of dedicated hardware and software. The costs associated with such installations have seen them largely restricted to the preserve of concert venues, cinemas, and institutions with the means to purchase and operate large-scale systems of this sort.</p>
<p>Advancements in embedded computing mean that there now exist an assortment of small, low-cost devices with support for audio Digital Signal Processing (DSP). These devices are relatively easy to program with open-source APIs and libraries, and may provide support for communication over ubiquitous computer networking equipment and protocols, an important capability in light of the rise of high-speed ethernet as a standard for multichannel audio transmission (<xref ref-type="bibr" rid="B7">Bakker et al., 2014</xref>). A network of such devices could be used to <italic>distribute</italic> the problem of audio spatialisation, permitting a modular, scalable approach that could lower the barrier to entry to what is otherwise a comparatively exclusive branch of audio research. This could in turn pave a more accessible path to research in a variety of domains in which sound field synthesis is a desirable component, including auditory scene personalisation (<xref ref-type="bibr" rid="B33">Geier et al., 2010</xref>), auralization and archaeoacoustics (<xref ref-type="bibr" rid="B9">Berger et al., 2023</xref>), telepresent videoconferencing (<xref ref-type="bibr" rid="B25">de Bruijn, 2004</xref>), and immersive networked music performance (<xref ref-type="bibr" rid="B61">Turchet and Tomasetti, 2023</xref>). Further, for implementations such as virtual acoustics, where computationally expensive impulse response convolutions play a part, a distributed system could afford an overall increase in available DSP resources.</p>
<p>In this article we describe the development of a distributed system for spatial and immersive audio. In <xref ref-type="sec" rid="s2">Section 2</xref> we discuss the technological and scholarly background to this project, including developments in networked audio and embedded hardware platforms, spatial audio techniques, and distributed audio systems. Building upon this background, and prior work on a microcontroller-based networked audio client (<xref ref-type="bibr" rid="B55">Rushton et al., 2023</xref>), <xref ref-type="sec" rid="s3">Section 3</xref> details the development of the proposed system, an overview of which is depicted in <xref ref-type="fig" rid="F1">Figure 1</xref>. With a view to optimising accessibility and interoperability with existing audio tools, the system&#x2019;s server component is encapsulated in an audio plugin suitable for use in a Digital Audio Workstation (DAW). The server delivers audio and control data to a network of microcontroller-based clients via a local area network, and clients, programmed with a parallelised spatial audio algorithm, use the audio and control data to perform their part of the distributed signal process. Minimising inter-client asynchronicity in a distributed audio context is the most significant technical problem, and the network client implementation incorporates strategies for addressing this challenge.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Overview of the proposed distributed, networked audio system. A general purpose computer runs DAW software, which in turn runs an instance of an audio plugin that encapsulates servers for audio and control data. The computer is physically connected, via a network switch, to microcontroller-based clients. The clients receive the audio and control data, which are delivered to a distributed audio spatialisation algorithm, and return an audio stream to the server (perhaps via some manner of signal processor). Clients deliver the output signals produced by their instance of the distributed algorithm to loudspeakers.</p>
</caption>
<graphic xlink:href="frvir-05-1391987-g001.tif"/>
</fig>
<p>The previous system was not formally evaluated, but, anecdotally, provided support for the holophonic effect of wave field synthesis. A perceptual evaluation of the new system was conducted, and this is described, along with a technical evaluation, in <xref ref-type="sec" rid="s4">Section 4</xref>. Finally, in <xref ref-type="sec" rid="s5">Section 5</xref>, we give an overview of our findings to date and describe our ambitions for future work.</p>
</sec>
<sec id="s2">
<title>2 Background</title>
<sec id="s2-1">
<title>2.1 Networked audio</title>
<p>The transmission of audio data has been a topic of research interest since the earliest days of computer networking as it is recognised today, i.e., over packet-switched networks, whereby data to be transmitted is grouped into packets&#x2014;or &#x201c;datagrams&#x201d; &#x2014; each consisting of a header and a payload. Voice transmission over ARPANET was being conducted as early as 1974 (<xref ref-type="bibr" rid="B58">Schulzrinne, 1992</xref>) and the first standard for voice communication over packet-switched networks&#x2014;the Network Voice Protocol (NVP) &#x2014; was released in 1977 (<xref ref-type="bibr" rid="B20">Cohen, 1977</xref>).</p>
<p>The NVP standard, with control messages for &#x2018;calling&#x2019; and &#x2018;ringing&#x2019;, was clearly intended for digital telephony, and communication was the primary focus of networked audio research well into the 1990s. Efforts on supporting real-time voice communication over wide area networks (WAN) centred on <italic>quality of service</italic> (QoS), particularly with regard to the perennial issues of latency, packet loss, and jitter&#x2014;inconsistencies in the rate of packet transmission (<xref ref-type="bibr" rid="B35">Hardman et al., 1995</xref>; <xref ref-type="bibr" rid="B36">1998</xref>). Work at this time dealt with streams of compressed audio data, and speech coding algorithms to overcome the deleterious effects of dropped packets over unreliable network paths and low-bandwidth connections.</p>
<p>Whereas the priority for digital telephony, and later voice over IP (VoIP) systems, is intelligibility, for musical purposes fidelity is of greater concern. The late 1990s, with the increasing availability of high-speed internet connections, saw the beginning of research into transmitting uncompressed audio data over the internet (<xref ref-type="bibr" rid="B18">Chafe et al., 2000</xref>; <xref ref-type="bibr" rid="B65">Xu et al., 2000</xref>). Work of this sort was spearheaded by the <italic>SoundWIRE</italic> project, developed by researchers at McGill University and Stanford University, and took the form of a wide variety of experiments with high quality audio transmission over both WAN and local area networks (LAN). These experiments included LAN-based real-time musical performances (<xref ref-type="bibr" rid="B18">Chafe et al., 2000</xref>), concert streaming over WAN (<xref ref-type="bibr" rid="B65">Xu et al., 2000</xref>; <xref ref-type="bibr" rid="B18">Chafe et al., 2000</xref>), and sonification of QoS via a distributed digital waveguide dubbed the <italic>Network Harp</italic> (<xref ref-type="bibr" rid="B18">Chafe et al., 2000</xref>; <xref ref-type="bibr" rid="B19">2002</xref>).</p>
<sec id="s2-1-1">
<title>2.1.1 Protocols and systems</title>
<p>VoIP research in the 1990s focused on audio codecs and data compression (<xref ref-type="bibr" rid="B62">Turletti, 1994</xref>; <xref ref-type="bibr" rid="B36">Hardman et al., 1998</xref>), seeking a compromise with the <italic>best-effort</italic> nature of internet service. The SoundWIRE project, in search of high audio quality, turned its attention directly to the basic transport layer protocols of the Internet Protocol suite: Transmission Control Protocol (TCP) and User Datagram Protocol (UDP). Chafe et al. characterised their compression-free system as taking a &#x201c;simplified approach&#x201d; to networked audio (<xref ref-type="bibr" rid="B18">Chafe et al., 2000</xref>), emphasising the importance of delivering multichannel audio of at least CD quality (16-bit, 44.1&#xa0;kHz) with as little latency as possible.</p>
<p>SoundWIRE experiments included TCP-based concert streaming. TCP&#x2019;s <italic>connection-oriented</italic>, one-to-one design enables packet flow control mechanisms that guarantee packet ordering and protect against packet loss (<xref ref-type="bibr" rid="B57">Schiavoni et al., 2013</xref>; <xref ref-type="bibr" rid="B5">AL-Dhief et al., 2018</xref>); at the expense of increased latency, these mechanisms safeguard quality of service, and thus audio fidelity; ideal for a remote concert scenario. UDP by comparison provides no such safeguards, but equally none of the associated computational or temporal overhead. Further, due to its <italic>connectionless</italic> model, <italic>many-to-many</italic> (multicast) and <italic>one-to-many</italic> (broadcast) modes of transmission are possible via address spaces reserved as part of the internet protocol standard (<xref ref-type="bibr" rid="B46">Meyer et al., 2010</xref>). Via UDP, SoundWIRE was able to run as a distributed digital waveguide over a WAN spanning around 4,500&#xa0;km (<xref ref-type="bibr" rid="B18">Chafe et al., 2000</xref>).</p>
<p>From the SoundWIRE project emerged <italic>JackTrip</italic> (<xref ref-type="bibr" rid="B13">C&#xe1;ceres and Chafe, 2010a</xref>; <xref ref-type="bibr" rid="B14">C&#xe1;ceres and Chafe, 2010b</xref>), a hybrid system that couples a TCP handshake with audio transmission over UDP, thus sidestepping the overhead of TCP packet flow control. Rather than relying on TCP&#x2019;s built-in mechanisms for stream integrity, JackTrip supplements UDP with a selection of optional buffering strategies that aim to optimise its operation in various network conditions. In this sense it is more flexible than TCP, but in effect JackTrip moulds UDP transmission into something akin to the connection-oriented model of TCP, and, in its &#x2018;hub server&#x2019; mode, into a kind of <italic>multiple one-to-one</italic> design&#x2014;multicast transmission is not possible.</p>
<p>UDP has emerged as the protocol of choice for platforms enabling remote musical collaboration, serving as the basis for NetJACK (<xref ref-type="bibr" rid="B15">Car&#xf4;t et al., 2009</xref>), part of the JACK Audio Connection Kit (a cross-platform audio host), the audiovisual performance streaming platform LOLA (<xref ref-type="bibr" rid="B28">Drioli et al., 2013</xref>), Jamulus (<xref ref-type="bibr" rid="B30">Fischer, 2015</xref>), Soundjack (<xref ref-type="bibr" rid="B54">Renaud et al., 2007</xref>), and other jamming-focused platforms, plus more recent entrants, the closed-source, but ultimately UDP-based networking component of Elk Audio OS (<xref ref-type="bibr" rid="B60">Turchet and Fischione, 2021</xref>), for instance. UDP also plays a fundamental role in networked media streaming, being the typical transport-layer protocol behind the Real-time Transport Protocol (RTP), and it features in proprietary networked audio systems such as Dante (Digital Audio Network Through Ethernet) (<xref ref-type="bibr" rid="B24">Dante, 2022</xref>).</p>
</sec>
<sec id="s2-1-2">
<title>2.1.2 AoE in the audio industry</title>
<p>In parallel with the work being carried out in academia on SoundWIRE, JackTrip and NetJACK, audio industry bodies&#x2014;the IEEE (Institute of Electrical and Electronics Engineers) and AES (Audio Engineering Society) standards groups, and companies like Audinate, the creators of Dante&#x2014;were taking an interest in networked audio. Traditional large-scale audio systems such as those used in broadcast, concert venues and recording studios rely on the installation of unwieldy combinations of analogue hardware and cabling, with many potential points of failure. Seeking literally to lighten the load posed by &#x201c;hundreds of kilograms&#x201d; (<xref ref-type="bibr" rid="B7">Bakker et al., 2014</xref>) of cabling in analogue audio installations, in the 2000s audio companies were looking to high speed ethernet as a means to simplify the provision of high-quality, multichannel audio in industry settings.</p>
<p>Key to these efforts was the release, in 2002, of the IEEE 1588 standard for the Precision Time Protocol (PTP), a means by which networked computer systems can achieve clock synchronicity (<xref ref-type="bibr" rid="B29">Edison et al., 2002</xref>). PTP (another protocol that typically uses UDP for transport) superseded the lower-resolution Network Time Protocol (NTP), and, under ideal conditions, can achieve synchronisation accuracy of sub-microsecond order (<xref ref-type="bibr" rid="B59">Tongzhou and Lunhui, 2022</xref>). Synchronisation is achieved via the exchange of timestamped packets, coupled with precise estimates for send and receive times. Precision is best when timestamps can be calculated at the <italic>physical layer</italic>&#x2014;layer 1 of the Open Systems Interconnection (OSI) model, of which the aforementioned transport layer (layer 3) is a component&#x2014;i.e., by dedicated timers at the level of the physical network interface. Legacy and low-cost networking equipment do not typically possess support for hardware timestamping, however (<xref ref-type="bibr" rid="B22">Correll and Barendt, 2005</xref>), and devices that do offer such support are markedly more expensive.<xref ref-type="fn" rid="fn1">
<sup>1</sup>
</xref> PTP can be deployed as a software-only implementation (<xref ref-type="bibr" rid="B22">Correll and Barendt, 2005</xref>), albeit with impaired accuracy and a protracted clock-convergence period.</p>
<p>Dante, with its promise of low-latency, highly-multichannel audio over wired LAN, and device synchronisation via hardware PTP, has become the <italic>de facto</italic> industry standard in networked audio (<xref ref-type="bibr" rid="B7">Bakker et al., 2014</xref>). Bakker et al. refer to Dante as an &#x201c;open&#x201d; system, which is true, perhaps, in the sense that companies can incorporate the Dante system into their products under licence from Audinate; from the perspective of the academic community, however, Dante is very much a closed-source initiative and not a suitable platform for research.</p>
<p>In 2011, IEEE released the Audio Video Bridging (AVB, IEEE 802.1) standard (<xref ref-type="bibr" rid="B38">IEEE, 2011</xref>), and AES67 followed in 2013 (<xref ref-type="bibr" rid="B37">Hildebrand, 2014</xref>). These open technical standards describe <italic>suites</italic> of protocols for tasks such as media transmission, device discovery and synchronisation, and interoperability with other systems. Both use PTP for device synchronisation; AES67 uses RTP for media transmission, whereas AVB uses the data-link layer (layer 2) Audio Video Transport Protocol. Open implementations of AVB and AES67 exist, but, being complex standards featuring many components, such implementations may not be complete, support for embedded platforms is limited,<xref ref-type="fn" rid="fn2">
<sup>2</sup>
</xref> and a reliance on PTP raises the barrier to entry. Ultimately, if an accessible solution is sought, attention must be turned back to the transport layer, and to UDP directly.</p>
</sec>
<sec id="s2-1-3">
<title>2.1.3 Challenges posed by networked audio</title>
<p>Time, especially when dealing with the fine margins posed by real-time audio processing, represents the principal source of difficulty in a networked audio setting.</p>
<p>
<italic>Jitter</italic> refers to fluctuations in the rate of transmission or processing. In a networked audio setting, jitter gives rise to a situation whereby the arrival of audio data does not correspond with the moments at which it is needed, and may be caused by a number of factors: packet prioritisation rules in the firmware of an ethernet switch, the timing of hardware interrupts for a computer&#x2019;s audio or networking subsystems, and software design decisions relating to network transmission or reception to name but three. In a naive implementation, jitter may result in a recipient either halting processing until it receives the expected data, or simply continuing without any data. In either case, the result is likely to be disruption of the integrity of the audio signal at the recipient in the form of audible discontinuities.</p>
<p>
<italic>Clock drift</italic> arises as an inevitable consequence of no source of time in a system of computation being perfectly uniform, and no two sources of time being identical. The timing of a computer system is typically governed by a crystal oscillator, whose operating frequency is subject to manufacturing tolerances, and whose stability is affected by factors such as ambient temperature, and computational load on the system it governs (<xref ref-type="bibr" rid="B45">Marouani and Dagenais, 2008</xref>). Relative drift, or <italic>skew</italic>, is the difference in clock rates between two or more systems. Whereas jitter is a transient phenomenon, clock drift is continuous, and as two distinct systems of time move in and out of phase with each other over the longer term, drift may indeed give rise to jitter.</p>
<p>In professional audio settings, devices may be synchronised via an authoritative clock source such as word clock, or, in a networked setting, via PTP. In the absence of such an authoritative source, e.g., over a wide area network, or if using hardware that does not support such measures, buffering strategies are typically employed, coupled with delay-locked loops and resampling (<xref ref-type="bibr" rid="B1">Adriaensen, 2005</xref>; <xref ref-type="bibr" rid="B2">Adriaensen, 2012</xref>).</p>
</sec>
</sec>
<sec id="s2-2">
<title>2.2 Hardware platforms</title>
<p>The notion of taking a distributed approach to DSP is reliant on the identification of a suitable supporting hardware platform. For an accessible, distributed audio application, the ideal computing platform should be small and inexpensive, plus easily and rapidly programmable; of course, it should also provide audio and networking hardware, and, ideally, well-documented APIs for programming and interacting with this hardware.</p>
<p>Recent years have seen the emergence of a number of small, low-cost platforms for embedded systems development, perhaps best known amongst these being the <italic>Arduino</italic> family of microcontroller development boards,<xref ref-type="fn" rid="fn3">
<sup>3</sup>
</xref> whose open-source Software Development Kit (SDK), software libraries, and Integrated Development Environment (IDE) have greatly improved the accessibility of development on embedded systems (<xref ref-type="bibr" rid="B48">Michon et al., 2020</xref>). Though support for audio is limited via Arduino devices, a number of audio-specific systems, programmable with the Arduino SDK and IDE, and operable with many Arduino-compatible add-ons (sensors, displays, etc.), have been produced; these include various <italic>ESP32</italic> and <italic>STM32</italic> models, and the <italic>Daisy</italic> and <italic>Teensy</italic> microcontroller ranges. These platforms benefit from the wealth of tools, documentation and support associated with Arduino and the surrounding D.I.Y. and maker communities. Also worthy of consideration are the <italic>Raspberry Pi</italic> and <italic>Bela</italic> platforms. Though these are <italic>Embedded Linux Systems</italic> rather than microcontrollers, they are small-footprint devices, suitable for embedded applications. Bela in particular has been designed with a focus on audio development and interaction via sensors; it can be programmed via a web-based IDE, and the user need not interact with the underlying Linux operating system. Raspberry Pi is less accessible as platform for embedded audio development, and tends to be operated as more of a general-purpose small computer, though support for treating the platform like a microcontroller&#x2014;taking a <italic>bare metal</italic> approach&#x2014;is offered via the <italic>Circle</italic> development environment.<xref ref-type="fn" rid="fn4">
<sup>4</sup>
</xref>
</p>
<p>The above systems are typically programmed in C&#x2b;&#x2b;, with support for audio development provided by libraries such as Daisy&#x2019;s <italic>DaisySP</italic> and Teensy&#x2019;s <italic>Teensy Audio Library</italic>, which each provide audio APIs and a selection of pre-made algorithms for audio synthesis and DSP. Bela, as an alternative to its C&#x2b;&#x2b; audio API, can be programmed with the graphical programming language PureData, and Teensy, as a complement to its Audio Library, offers a web-based <italic>Audio System Design Tool</italic>, via which the user may describe an audio system diagrammatically and export the result to C&#x2b;&#x2b;.</p>
<p>This profusion of tools and platform-specific APIs can render embedded audio development somewhat difficult to approach. A concerted effort has been made, however, by the community behind the <italic>Faust</italic> programming language,<xref ref-type="fn" rid="fn5">
<sup>5</sup>
</xref> to provide support for embedded platforms. Faust is a functional paradigm, audio domain-specific language, that was created to serve as a &#x201c;viable and efficient alternative to C/C&#x2b;&#x2b;&#x201d; (<xref ref-type="bibr" rid="B52">Orlarey et al., 2009</xref>) for the development of audio applications on a variety of platforms. In Faust, a user can write high level sound synthesis or DSP code and export the result to C&#x2b;&#x2b; that meets the requirements of the audio API on a given target platform. This is achieved via a series of platform-specific &#x201c;architecture files&#x201d; and Faust&#x2019;s <monospace>faust2[&#x2026;]</monospace> tools,<xref ref-type="fn" rid="fn6">
<sup>6</sup>
</xref> which include <monospace>faust2bela</monospace>, <monospace>faust2teensy</monospace>, etc. (<xref ref-type="bibr" rid="B47">Michon et al., 2019</xref>; <xref ref-type="bibr" rid="B48">2020</xref>). Developers are thus able to focus on writing audio code, rather than being concerned with the peculiarities of the device or system upon which they wish to deploy their program; further, Faust&#x2019;s support for a variety of embedded platforms facilitates testing and rapid prototyping.</p>
<p>A comparison of selected devices can be found in <xref ref-type="table" rid="T1">Table 1</xref>. Bela is significantly more powerful than the microcontroller systems, but it is commensurately costly. The Raspberry Pi is also very capable, and a model with 1&#xa0;GB RAM may cost as little as &#x20ac;30; its operating system stands as an impediment, however, to implementations that seek to prioritise audio functionality above all. Support for bare metal development on Raspberry Pi is not comprehensive, and there is no Faust tool to produce code that is compatible with Circle. Daisy Seed is well-appointed with memory (which is important for DSP algorithms featuring long delay-lines, for example,), but does not provide ethernet support. Teensy 4.1, and the selected ESP32 and STM32 devices support networking via ethernet add-ons, but the ESP32&#x2019;s CPU is underpowered, and the STM32 is unfavourably-priced. Though lacking in memory, Teensy&#x2019;s processor, low price, and networking support make it an attractive candidate platform for a distributed, networked audio implementation. Further, thanks to the presence of a vibrant developer community, utilities such as <italic>TyTools</italic>
<xref ref-type="fn" rid="fn14">
<sup>14</sup>
</xref> exist, and can be used to program multiple Teensy devices in a single command&#x2014;useful for a system distributed amongst many such devices.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Comparison of selected embedded audio development platforms. Prices as of January 2024.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Platform</th>
<th align="center">Processor</th>
<th align="center">Memory</th>
<th align="right">Price</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Teensy 4.1<xref ref-type="fn" rid="fn7">
<sup>7</sup>
</xref>
</td>
<td align="center">ARM Cortex-M7 600&#xa0;MHz</td>
<td align="center">1&#xa0;MB SDRAM</td>
<td align="right">&#x20ac;32</td>
</tr>
<tr>
<td align="center">Daisy Seed<xref ref-type="fn" rid="fn8">
<sup>8</sup>
</xref>
</td>
<td align="center">ARM Cortex-M7 480&#xa0;MHz</td>
<td align="center">64&#xa0;MB SDRAM</td>
<td align="right">&#x20ac;28</td>
</tr>
<tr>
<td align="center">ESP32-LyraTD<xref ref-type="fn" rid="fn9">
<sup>9</sup>
</xref>
</td>
<td align="center">Dual core Xtensa LX6 240&#xa0;MHz</td>
<td align="center">8&#xa0;MB PSRAM</td>
<td align="right">&#x20ac;19</td>
</tr>
<tr>
<td align="center">STM32H747I<xref ref-type="fn" rid="fn10">
<sup>10</sup>
</xref>
</td>
<td align="center">ARM Cortex-M7 480&#xa0;MHz &#x2b; M4 240&#xa0;MHz</td>
<td align="center">1&#xa0;MB RAM</td>
<td align="right">&#x20ac;94</td>
</tr>
<tr>
<td align="center">Bela<xref ref-type="fn" rid="fn11">
<sup>11</sup>
</xref>
</td>
<td align="center">ARM Cortex-A8 1&#xa0;GHz<xref ref-type="fn" rid="fn12">
<sup>12</sup>
</xref>
</td>
<td align="center">512&#xa0;MB SDRAM</td>
<td align="right">&#x20ac;190</td>
</tr>
<tr>
<td align="center">Raspberry Pi 4<xref ref-type="fn" rid="fn13">
<sup>13</sup>
</xref>
</td>
<td align="center">ARM Cortex-A72 1.8&#xa0;GHz</td>
<td align="center">1&#xa0;GB&#x2013;8&#xa0;GB SDRAM</td>
<td align="right">&#x20ac;30&#x2013;100</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>One respect in which Teensy is found wanting is audio fidelity. By default, its audio add-on (or <italic>shield</italic>) produces CD quality output (16-bit, 44.1&#xa0;kHz), falling short of modern requirements for high-quality audio, such as offered by Daisy Seed (24-bit, 96&#xa0;kHz). While Teensy&#x2019;s sampling rate can be increased, sample resolution is fixed at the level of the device&#x2019;s audio codec. In spite of this shortcoming, and in light of its other, more advantageous qualities, Teensy was selected as the platform upon which to conduct development.</p>
</sec>
<sec id="s2-3">
<title>2.3 Audio spatialisation</title>
<p>Audio spatialisation is, plainly put, the practice of distributing sound in space. The spatialisation of <italic>primary sound sources</italic>, e.g., sound captured by microphones, stored as digital audio files, or synthesised in real-time, can be achieved simply by delivering those primary sources to <italic>secondary sound sources</italic>, i.e., loudspeakers or headphones. Exploiting auditory cues, and the nature of the propagation of sound, it is possible to suggest the presence of primary sources at arbitrary locations, independent of the secondary source distribution. The motivation behind audio spatialisation, then, is to create (or indeed <italic>recreate</italic>) sonic environments for creative and immersive purposes, such as for virtual reality experiences, in cinematic settings, for music production or art installations, to give but a handful of examples.</p>
<p>A number of techniques exist for what is termed <italic>sound field synthesis</italic> (<xref ref-type="bibr" rid="B3">Ahrens, 2012</xref>; <xref ref-type="bibr" rid="B51">Nicol, 2017</xref>), all of which essentially take the form of applying some manner of <italic>driving function</italic> to an input audio signal to generate an appropriate driving signal to be delivered to a secondary sound source in the listening environment (<xref ref-type="bibr" rid="B3">Ahrens, 2012</xref>). For a loudspeaker at position <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mtable class="matrix">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mi>x</mml:mi>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mi>y</mml:mi>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mi>z</mml:mi>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, the time-domain driving signal <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> can be expressed as a convolution of the input signal <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and the driving function <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>:<disp-formula id="e1">
<mml:math id="m5">
<mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mspace width="2.77695pt" class="tmspace"/>
<mml:mo>&#x2217;</mml:mo>
<mml:mspace width="2.77695pt" class="tmspace"/>
<mml:mi>d</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <inline-formula id="inf5">
<mml:math id="m6">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes time.</p>
<p>Commonly-employed approaches to sound field creation can be grouped into two broad categories: amplitude- and time-based panning techniques, and physical sound field recreation approaches.</p>
<sec id="s2-3-1">
<title>2.3.1 Periphony and binaural reproduction</title>
<p>The former, <italic>periphonic</italic>, types encompass stereophony and surround-sound systems, consisting of secondary sources in a planar arrangement equidistant from the listening position. These techniques exploit the interaural level difference (ILD) cue, i.e., the difference in perceived amplitude relative to the listener&#x2019;s ears (<xref ref-type="bibr" rid="B53">Pulkki, 1997</xref>; <xref ref-type="bibr" rid="B63">Verheijen, 1998</xref>; <xref ref-type="bibr" rid="B66">Ziemer, 2020</xref>), to encourage the listener to localise sound to a position on the circumference of an arc or circle around the listening position. For systems of this sort, the driving function is a constant scalar value, or, for a moving phantom source, a time-varying function that returns a scalar value. Such periphonic approaches can extend to three dimensions in the case of vector base amplitude panning (VBAP) (<xref ref-type="bibr" rid="B53">Pulkki, 1997</xref>), which uses trios of speakers to position phantom sources on the surface of a sphere with the listening position at its origin.</p>
<p>Time-based panning effects, by contrast, make use of the interaural time difference (ITD) cue to give the impression of a phantom source located toward the loudspeaker producing the signal at the earliest time (<xref ref-type="bibr" rid="B53">Pulkki, 1997</xref>; <xref ref-type="bibr" rid="B63">Verheijen, 1998</xref>). Thus the driving function for a time-based panning system is a delay of the form:<disp-formula id="e2">
<mml:math id="m7">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf6">
<mml:math id="m8">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the duration of the delay.</p>
<p>The effects of ILD and ITD cues transfer to headphone-based listening, in which case, rather than periphonic, they form a sort of <italic>in-head</italic> localisation (<xref ref-type="bibr" rid="B3">Ahrens, 2012</xref>). For a significantly more naturalistic auditory outcome, ILD and ITD cues, when combined with filters describing the dispersive and absorptive effects of the head, torso and outer ears, constitute a Head-Related Transfer Function (HRTF), the key component in what is termed <italic>binaural reproduction</italic>. Binaural recordings are taken either with a dummy head or ear-mounted microphones, and thus the signal for each ear is coloured by the head used during recording, or HRTF measurements can be taken and used to describe filters to be applied to arbitrary signals at playback. Binaural sound is suited to headphone-based listening but may be achieved with loudspeakers if suitable cross-talk cancellation is applied (<xref ref-type="bibr" rid="B40">Kaiser, 2011</xref>).</p>
<p>Periphonic approaches are subject to the phenomenon of an ideal listening position, or <italic>sweet-spot</italic> (<xref ref-type="bibr" rid="B51">Nicol, 2017</xref>), that is a listening position away from which the spatialisation effect is significantly degraded. Binaural reproduction, if not coupled with head motion tracking, is similarly afflicted by an ideal position and orientation (<xref ref-type="bibr" rid="B63">Verheijen, 1998</xref>); further, for faithful reproduction, HRTFs should be individualised (<xref ref-type="bibr" rid="B26">De Poli and Rocchesso, 1998</xref>). As such, these techniques are not suited to collective and immersive auditory experiences whereby multiple participants may move freely about their environment.</p>
</sec>
<sec id="s2-3-2">
<title>2.3.2 Physically-inspired techniques</title>
<p>Physical approaches fall into two main types: wave field synthesis (WFS) (<xref ref-type="bibr" rid="B11">Berkhout et al., 1993</xref>) and ambisonics (and higher-order ambisonics&#x2014;HOA) (<xref ref-type="bibr" rid="B31">Frank et al., 2015</xref>). Rather than directly manipulating sound localisation cues, these types seek to trigger those cues indirectly by synthesising a sound field as if it had been created by &#x201c;true&#x201d; acoustic sources.</p>
<p>In the case of ambisonics, the sound field is decomposed into &#x201c;spherical harmonics&#x201d;, spatial functions described by linear sums of directional components of increasing order (<xref ref-type="bibr" rid="B51">Nicol, 2017</xref>). Like periphonic approaches, ambisonics suffers from a sweet-spot effect which worsens with attempts to reproduce sounds of higher frequency, but can be mitigated by reproducing higher-order modes and increasing the density of the distribution of secondary sources.</p>
<p>WFS is based upon Huygens&#x2019; principle, originating in the field of optics, which states that a propagating wavefront can be recreated by a distribution of secondary point sources (<xref ref-type="bibr" rid="B50">Mueller, 1971</xref>; <xref ref-type="bibr" rid="B11">Berkhout et al., 1993</xref>; <xref ref-type="bibr" rid="B8">Belloch et al., 2021</xref>) (see <xref ref-type="fig" rid="F2">Figure 2</xref>). WFS is variously termed a form of <italic>acoustic holography</italic> or <italic>holophony</italic> (<xref ref-type="bibr" rid="B10">Berkhout, 1988</xref>; <xref ref-type="bibr" rid="B3">Ahrens, 2012</xref>). Effectively, by timing the reproduction of an input signal at an array of secondary sources, a wavefront associated with a virtual sound source can be synthesised. To simulate auditory cues related to perceived distance, a filter can be applied to model losses to the virtual medium of acoustic propagation. The principle assumes a continuous array of secondary sources but of course in practice it is necessary to use a discrete array of loudspeakers, which, much as is the case with HOA, has consequences for spatial resolution; to mitigate the issue of spatial aliasing, whereby sounds of higher frequency cannot be recreated unambiguously (Winter et al., 2018), secondary sources should be placed very close together. Consequently, to serve a large listening area, many speakers, and thus many audio channels, are required.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Holophony. Huygens&#x2019; principle states that the propagation of a wavefront can be recreated by a collection of secondary point sources. The bottom of the figure represents a virtual sound field, and the top a real sound field, separated by a row of secondary point sources (loudspeakers). The small circle represents a virtual sound source and the dashed arcs are virtual wavefronts associated with that sound source; the small solid arcs are wavefronts produced by the array of secondary point sources; the large solid arcs represent the propagation of a reconstructed wavefront in the real sound field.</p>
</caption>
<graphic xlink:href="frvir-05-1391987-g002.tif"/>
</fig>
<p>Via appropriate timing of the delivery of a primary sound source to the secondary source array, it is possible to synthesise virtual sound sources, plane waves, and <italic>focused</italic> sound sources, corresponding with concave, flat, and convex synthesised wavefronts respectively; the latter, dependent on the location of the listener, appear to emanate from within the real sound field, rather than its virtual counterpart.</p>
<p>Focusing on the former kind, however, for <inline-formula id="inf7">
<mml:math id="m9">
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> virtual sources, the time-domain driving signal <inline-formula id="inf8">
<mml:math id="m10">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> for the secondary source at <inline-formula id="inf9">
<mml:math id="m11">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> may be expressed as a sum of input-signal driving-function convolutions (see <xref ref-type="disp-formula" rid="e1">Equation 1</xref>):<disp-formula id="e3">
<mml:math id="m12">
<mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where the driving function <inline-formula id="inf10">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is (<xref ref-type="bibr" rid="B3">Ahrens, 2012</xref>):<disp-formula id="e4">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>This is, in effect, a physically-informed extension of <xref ref-type="disp-formula" rid="e2">Equation 2</xref>. The <italic>WFS prefilter</italic> <inline-formula id="inf11">
<mml:math id="m15">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is a function that simulates the absorption of energy into the simulated medium of acoustic propagation. The delta function <inline-formula id="inf12">
<mml:math id="m16">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> has the effect of delaying the prefilter, and thus <inline-formula id="inf13">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, by the time of propagation for a medium with propagation speed <inline-formula id="inf14">
<mml:math id="m18">
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (typically modelled as 343&#xa0;m/s for sound in air). The components of the driving function are depicted in <xref ref-type="fig" rid="F3">Figure 3</xref>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>The driving signal for the WFS secondary source at position <inline-formula id="inf15">
<mml:math id="m19">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, for virtual primary source <inline-formula id="inf16">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, is dependent on the distance <inline-formula id="inf17">
<mml:math id="m21">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> of the primary source from the secondary source. This corresponds with a propagation delay via the simulated medium of propagation, coupled with a filter describing losses to that medium.</p>
</caption>
<graphic xlink:href="frvir-05-1391987-g003.tif"/>
</fig>
</sec>
<sec id="s2-3-3">
<title>2.3.3 State of the art spatial audio installations</title>
<p>As described, for optimal spatial resolution, systems implementing ambisonics and WFS require many output channels; in effect, the more channels, and the greater the loudspeaker-density, the better.</p>
<p>The Multisensory Experience Lab at Aalborg University (AAU), Copenhagen, hosts a 64-channel, square-array WFS system covering an area of 4&#xa0;m <inline-formula id="inf18">
<mml:math id="m22">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 4&#xa0;m (<xref ref-type="bibr" rid="B34">Grani et al., 2016</xref>). It is driven by a Mac Pro desktop computer, connected, via a USB MADI (Multichannel Audio Digital Interface) interface, to two 32-channel MADI to analogue converters. Though this system&#x2019;s WFS engine is provided by open-source <italic>WFSCollider</italic> software,<xref ref-type="fn" rid="fn15">
<sup>15</sup>
</xref> being a centralised system, the computer that co-ordinates its operation is powerful &#x2014; 12-core CPU, 64&#xa0;GB RAM&#x2014;and was costly at the time of purchase; likely on the order of several thousand Euros. With further regard to cost, the current equivalent MADI to analogue converter models are priced at roughly &#x20ac;5,000 apiece. Excluding the cost of loudspeakers (and the gantry upon which they are mounted) one may reasonably place an estimate of &#x20ac;250 per output channel on this system.</p>
<p>The world&#x2019;s largest dedicated WFS system, at TU Berlin, features over 800 output channels served by a distributed cluster of fifteen computers acting as audio nodes and two additional control computers (<xref ref-type="bibr" rid="B6">Baalman et al., 2007</xref>).<xref ref-type="fn" rid="fn16">
<sup>16</sup>
</xref> The fifteen audio nodes of the TU Berlin system are each equipped with a MADI audio interface, connected to a MADI to ADAT bridge; these can, at the time of writing, be purchased for around &#x20ac;1,400 and &#x20ac;4,500 each, respectively, for a total of &#x20ac;88 500. Imagining &#x20ac;2000 per computer, a conservative estimate of &#x20ac;150/channel may be reached, however the expense associated with this system is difficult to assess as it is tied to the concert hall space in which it resides. Indeed, one thing besides expense that unites the systems at AAU and TU Berlin, plus the 339-speaker hybrid WFS/ambisonics system at IRCAM (Paris)<xref ref-type="fn" rid="fn17">
<sup>17</sup>
</xref>, the 512-channel WFS system at the Rensselaer Polytechnic Institute (NY, United States)<xref ref-type="fn" rid="fn18">
<sup>18</sup>
</xref>, and the behemoth installation at the <italic>Sphere</italic> (Las Vegas, United States)<xref ref-type="fn" rid="fn19">
<sup>19</sup>
</xref> is their <italic>in-situ</italic> nature; these are site-specific systems with, at best, limited flexibility.</p>
<p>What one is afforded by WFS installations of this scale and expense, however, is high quality audio reproduction with tantamount to perfect output synchronicity; the integrity of the holophonic effect of WFS is, unavoidable matters of spatial aliasing aside, guaranteed.</p>
</sec>
</sec>
<sec id="s2-4">
<title>2.4 Distributed audio systems</title>
<p>As alluded to earlier in this section, our aim is to distribute the problem of audio spatialisation, and it is worthwhile to revisit why this is the case. In the broadest terms, a distributed system is <italic>&#x201c;a collection of independent entities that cooperate to solve a problem that cannot be individually solved&#x201d;</italic> (<xref ref-type="bibr" rid="B42">Kshemkalyani and Singhal, 2011</xref>). An ideal distributed system is characterised by: <italic>modularity</italic>, being comprised of separate, interchangeable entities; <italic>scalability</italic>, being extensible without incurring a performance penalty to the system as a whole, and; <italic>improved performance/cost ratio</italic>, since it can be constructed to meet the proportion that circumstances require, with the minimum degree of redundancy. Additionally, for algorithms that can be effectively <italic>parallelised</italic>, a distributed system may provide more aggregate computational power than its centralised counterpart.</p>
<p>Where a distributed system may suffer, by contrast, is in terms of reliability. Nodes in a distributed computational system must be served with power and access to the data they require in order to operate, which entails a proliferation of potential points of failure. The other side to the coin of modularity is a concern regarding the programmability of such a system; ensuring that all entities possess up-to-date instructions for operation may not be trivial. Further, some algorithms may be better suited to parallelisation than others; efficient use of increased computational resources is not guaranteed.</p>
<p>Distributed audio processing is by no means a matter without precedent. (Indeed, the WFS installation at TU Berlin described in <xref ref-type="sec" rid="s2-3-3">Section 2.3.3</xref> is of course a distributed system, albeit not an especially accessible one.) A selection of prior work in distributed DSP and audio spatialisation, plus systems incorporating microcontrollers and single-board computers is detailed below.</p>
<sec id="s2-4-1">
<title>2.4.1 State of the art distributed audio systems</title>
<p>Applications of SoundWIRE to what its creators termed <italic>Internet Acoustics</italic> (<xref ref-type="bibr" rid="B19">Chafe et al., 2002</xref>) clearly stand as examples of distributed audio processing. These include a network reverberator (<xref ref-type="bibr" rid="B16">Chafe, 2018</xref>), or <italic>&#x201c;transcontinental echo chamber&#x201d;</italic> (<xref ref-type="bibr" rid="B18">Chafe et al., 2000</xref>), plus the aforementioned <italic>Network Harp</italic>. Experiments of this sort were intended initially as sonifications of QoS&#x2014;a characteristic of network systems that is difficult to represent in real time in graphical or textual form due to the ephemeral nature of the phenomena of jitter and packet loss&#x2014;but stand as fascinating applications in their own right of digital audio in the age of computer networking. Subsequent work on JackTrip has focused on optimising networked audio less as a creative tool in itself, and more in service of the social and communal aspects of music participation and appreciation in a networked world, topics that came to the fore in computer music research during the COVID-19 pandemic (<xref ref-type="bibr" rid="B12">Bosi et al., 2021</xref>; <xref ref-type="bibr" rid="B56">Sacchetto et al., 2021</xref>). That being said, more recent work on Internet Acoustics in an embedded context has yielded a port of JackTrip to the Raspberry Pi (<xref ref-type="bibr" rid="B17">Chafe and Oshiro, 2019</xref>).</p>
<p>Examples of distributed music production systems include the work of Lago and Kon (<xref ref-type="bibr" rid="B43">Lago and Kon, 2003</xref>), whose UDP-based system featured clients that acted as delegates for audio processing, and <xref ref-type="bibr" rid="B32">Gabrielli et al. (2012)</xref>, who demonstrated a wireless relay of audio and control-data processors. Latency was an important metric for the latter system, and the authors measured latency via transmission round-trip times using a low frequency sawtooth wave as a timer (see <xref ref-type="sec" rid="s4-1">Section 4.1</xref> for an application of this technique).</p>
<p>Distributed approaches to audio spatialisation include Lopez-Lezcano&#x2019;s &#x201c;network sound card&#x201d; (<xref ref-type="bibr" rid="B44">Lopez-Lezcano, 2012</xref>), and embedded implementations as described by <xref ref-type="bibr" rid="B27">Devonport and Foss, (2019)</xref> and <xref ref-type="bibr" rid="B8">Belloch et al. (2021)</xref>. The latter two address aims closely aligned with the work described here, but are based on costly computing platforms. Devonport and Foss used AVB, and thus PTP, for synchronisation; Belloch et al. employed a GPU-based hardware platform, reporting client synchronisation to the millisecond range&#x2014;likely not sufficient for timing-critical audio spatialisation effects.</p>
<p>Also of interest is the OTTOsonics project (<xref ref-type="bibr" rid="B49">Mitterhuber et al., 2022</xref>); its emphasis on a fully-costed, flexible, do-it-yourself alternative to conventional spatial audio systems is pertinent to this work, though it diverges in its use of AVB, and associated hardware for audio transmission. A full 24-channel OTTOsonics system, including speakers and audio interface, is costed at around &#x20ac;2,600 (&#x20ac;108.33/channel), however, which certainly places it favourably when compared with state-of-the-art spatialisation systems.</p>
</sec>
</sec>
</sec>
<sec sec-type="methods" id="s3">
<title>3 Methods</title>
<p>As illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref> (page 2), the proposed system is distributed across distinct computing platforms (a general purpose computer; a network of microcontrollers), and software elements serving a variety of purposes (server and client instances for transmission and reception of networked audio and control data, plus a DSP algorithm). In the subsections that follow, these elements are described in detail; finally, in <xref ref-type="sec" rid="s3-4">Section 3.4</xref>, an overview of the system and its operation is provided.</p>
<sec id="s3-1">
<title>3.1 The networked audio server</title>
<p>TCP is, as described in <xref ref-type="sec" rid="s2-1-1">Section 2.1.1</xref>, a connection-based, one-to-one protocol, so the JackTrip connection model enforces a sort of pseudo-connectionfulness on the otherwise connectionless UDP. The result is a system which permits only unicast UDP transmission, and, for multiple clients, must send a duplicate of the outgoing stream of audio datagrams to each connected client. A JackTrip server creates a sender and a receiver task for each client that connects (<xref ref-type="bibr" rid="B13">C&#xe1;ceres and Chafe, 2010a</xref>); notionally this entails, should enough clients connect, exhaustion of all available network bandwidth; as such, a unicast system does not meet the requirement of scalability as described in <xref ref-type="sec" rid="s2-4">Section 2.4</xref>.</p>
<p>A multicast NetJACK server was considered, but creating a client implementation on what is essentially a bare-metal platform in the shape of the Teensy, was not practical. Further, due to a break in compatibility with Mac OS X systems, JACK-based approaches are not truly cross-platform.<xref ref-type="fn" rid="fn20">
<sup>20</sup>
</xref> Prioritising simplicity, in the form of an audio server with minimal dependencies and a very specific task to achieve, we embarked upon the design of a bespoke multicast networked audio server.</p>
<sec id="s3-1-1">
<title>3.1.1 Designing a networked audio protocol</title>
<p>Dependent on the intended application, and if assumptions can be made about matters such as sampling rate and bit resolution, a <italic>no-protocol</italic> approach, such as described by Lopez-Lezcano (<xref ref-type="bibr" rid="B44">Lopez-Lezcano, 2012</xref>), may be a viable one. To improve the flexibility of the system and render it somewhat future-proof, however, a simple packet header was devised. Its structure is given in <xref ref-type="statement" rid="listing_1">Listing 1</xref>.</p>
<p>The resulting six-byte header comprises a two-byte (unsigned 16-bit integer) packet sequence number, to be incremented by the sender, plus four further bytes describing the structure of the audio data in the packet. Commonly-encountered sampling rates, and buffer sizes greater than 255, cannot be represented by unsigned eight-bit integers, so these are supported by enumerations inspired by those used by JackTrip.<xref ref-type="fn" rid="fn21">
<sup>21</sup>
</xref>
</p>
<p>
<statement content-type="algorithm" id="listing_1">
<label>Listing 1. Packet header structure.</label>
<p>
<inline-graphic xlink:href="frvir-05-1391987-fx1.tif"/>
</p>
<p>
<monospace>BufferSize</monospace> describes the number of audio frames per packet<xref ref-type="fn" rid="fn22">
<sup>22</sup>
</xref> as the <inline-formula id="inf19">
<mml:math id="m23">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>th power of 2; for example, the enumeration <monospace>BufferSizeT</monospace> features a member <monospace>BufferSizeT::BUF16</monospace>; 16 being the fourth power of 2, this member is assigned the number 4. The <monospace>BitResolution</monospace> field could be used to transmit one of 8, 16, 24, or 32 as-is; there is a utility, however, when decoding a packet, in knowing the number of <italic>bytes</italic> per audio sample, so this is the number that is represented, e.g., <monospace>BitResolutionT::BIT16</monospace> takes the value 2, the number of bytes in a 16-bit integer.</p>
<p>For well-formed packets, <monospace>BufferSize</monospace> could be inferred from the size of the packet (minus its header), divided by <monospace>NumChannels</monospace> and <monospace>BitResolution</monospace>. To permit scope for the detection of malformed packets, however, the expense of an additional byte in the header was deemed a reasonable one. Finally, the sequence number is intended as a means for a recipient to identify the occurrence of packet loss, and will wrap around to zero every 65,536 packets.</p>
<p>One piece of information that is not stated in the packet header is the manner in which audio samples in the packet should be interleaved. The assumption taken&#x2014;indeed, the same assumption used by JackTrip&#x2014;is that audio data is channel-interleaved, i.e., audio data consists of a contiguous block of samples for one channel, followed by a block for the next channel, and so on.</p>
</statement>
</p>
</sec>
<sec id="s3-1-2">
<title>3.1.2 Server design</title>
<p>The networked audio server was written in C&#x2b;&#x2b; using utility classes provided by the JUCE framework for the development of audio applications<xref ref-type="fn" rid="fn23">
<sup>23</sup>
</xref> and is encapsulated as a class called <monospace>NetAudioServer</monospace>. Initial development was conducted on a basic console application, and later work targeted a DAW plugin comprising a consolidated audio server and wave field synthesis controller.</p>
<p>The <monospace>NetAudioServer</monospace> instance expects to receive blocks of multichannel audio from an audio application&#x2019;s main processing loop. It sets up network <italic>sender</italic> and <italic>receiver</italic> execution threads, and assigns a network socket to each; a socket is essentially a numerical identifier for an <italic>&#x201c;endpoint for [network] communication&#x201d;</italic> (<xref ref-type="bibr" rid="B41">Kerrisk, 2023</xref>) to which a type&#x2014;on Linux systems, <monospace>SOCK_STREAM</monospace> for TCP, <monospace>SOCK_DGRAM</monospace> for UDP&#x2014;can be assigned. To avoid potentially blocking the audio application&#x2019;s main processing thread with networking operations, upon receiving an audio block the server writes it to an intermediate buffer&#x2014;a first-in-first-out (FIFO) structure&#x2014;and signals the sender thread that a block is ready for transmission. The sender thread, which as been awaiting such a signal, then requests samples from the FIFO; these are stored as contiguous channels of 32-bit floating point samples and converted, when requested, to the bit resolution specified in a packet header created when <monospace>NetAudioServer</monospace> is initialised. Byte order, or <italic>endianness</italic> (<xref ref-type="bibr" rid="B21">Cohen, 1981</xref>), is also specified as part of this conversion. Though network byte order is typically big-endian, or Most Significant Byte (MSB) first, it was found that little-endian transmission meant that samples could be decoded trivially at the client side.<xref ref-type="fn" rid="fn24">
<sup>24</sup>
</xref> Upon receiving the requested samples, the sender thread writes these to its socket, which has been configured to connect to a UDP multicast group. This process is illustrated in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Overview of operation of the networked audio server. The network sender awaits notification of readiness to read samples from a first-in-first-out buffer of audio samples. The audio processor receives audio channels from a multichannel source (e.g., a DAW); at each iteration of its processing loop, it writes samples to the FIFO; upon write-completion, the FIFO sends a signal to the network sender that a block of samples is ready. Samples are converted to the desired bit resolution and byte order and bundled into a UDP packet which is then written to the network.</p>
</caption>
<graphic xlink:href="frvir-05-1391987-g004.tif"/>
</fig>
<p>
<statement content-type="algorithm" id="listing_2">
<label>Listing 2. Network capture: ethernet frame containing a UDP audio packet.</label>
<p>
<inline-graphic xlink:href="frvir-05-1391987-fx2.tif"/>
</p>
</statement>
</p>
<p>
<xref ref-type="statement" rid="listing_2">Listing 2</xref> shows an example network capture of an outgoing audio packet. Bytes <monospace>0x0000</monospace> to <monospace>0x0029</monospace> comprise the headers for the data link (ethernet), network (IPv4), and transport (UDP) OSI layers including the destination address: at position <monospace>0x001e</monospace>, the bytes <monospace>0xe004e004</monospace>, or <monospace>224.4.224.4</monospace>, a valid (and unassigned) UDP multicast address from the second <italic>ad hoc</italic> address block as specified in the IANA multicast address assignment guidelines (<xref ref-type="bibr" rid="B46">Meyer et al., 2010</xref>). The six subsequent bytes are the header inserted into the packet by <monospace>NetAudioServer</monospace>. In <xref ref-type="statement" rid="listing_2">Listing 2</xref> these are:<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf20">
<mml:math id="m24">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <monospace>0x1cdf</monospace>: a sequence number (little-endian) of 7391<sub>10</sub>;<xref ref-type="fn" rid="fn25">
<sup>25</sup>
</xref>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf21">
<mml:math id="m25">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <monospace>0x04</monospace>: buffer size 4 corresponding with <monospace>BufferSizeT::BUF16</monospace>;</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf22">
<mml:math id="m26">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <monospace>0x02</monospace>: sampling rate 2 corresponding with <monospace>SamplingRateT::SR44</monospace>;</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf23">
<mml:math id="m27">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <monospace>0x02</monospace>: bit resolution 2 corresponding with <monospace>BitResolutionT::BIT16</monospace>;</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf24">
<mml:math id="m28">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <monospace>0x02</monospace>: 2 audio channels.</p>
</list-item>
</list>
</p>
<p>Audio data begins at byte <monospace>0x0030</monospace>. Since the header indicates that there are two channels of 16-bit audio, and a buffer size of 16 frames, it is clear that the data for channel 1 encompasses the 32 bytes from <monospace>0x0030</monospace> to <monospace>0x004f</monospace>, and channel 2 the remaining bytes.</p>
<p>Here, channel 1 is a test signal, a unit amplitude-increment unipolar sawtooth wave, i.e., a signal whose amplitude starts at zero, and increments by 1 at each sample until it reaches the maximum value that a signed 16-bit integer may take &#x2014; 32 767<sub>10</sub> &#x2014; at which point it wraps around to zero and repeats. This test signal serves two important purposes. First, its impulse-like behaviour once every 32,768 samples (roughly .74&#xa0;s at a sampling rate of 44.1&#xa0;kHz) is useful for taking basic synchronicity measurements, e.g., involving connecting two clients&#x2019; audio outputs to an oscilloscope. Second, this numerically-predictable signal serves as a means to inspect the integrity of the audio server algorithm, and to verify that the expected sample interleaving and endianness is employed. Inspecting the first sixteen samples of the first audio channel it is evident that the amplitude values increment on a per-sample basis, and, since it is the first byte that increases with each sample, that samples are transmitted little-endian.</p>
<p>The purpose of the server&#x2019;s receiver thread is to poll its socket for traffic reaching the multicast group from connected clients. Clients are programmed to return a stream of packets of audio data to the multicast group, and the receive thread uses the existence of a such a stream, with a given origin IP address, to indicate the presence of a client at that address. If a client fails to return a packet for more than 1&#xa0;s it is considered disconnected. Clients could announce their presence with any periodic UDP transmission, but the possibility of returning audio data facilitates the measurement of client synchronicity via transmission round trip times (see <xref ref-type="sec" rid="s4-1">Section 4.1</xref>).</p>
</sec>
<sec id="s3-1-3">
<title>3.1.3 Transmission considerations</title>
<p>Ethernet frames, and UDP datagrams by extension, are subject to size limitations. The maximum transmissible unit (MTU) of a transport medium is the limit on the size of a packet that can be sent without fragmentation, i.e., without being split into multiple sub-packets. Two bytes are allocated to the &#x2018;Total Length&#x2019; field of the IPv4 header, which suggests an MTU of <inline-formula id="inf25">
<mml:math id="m29">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>16</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>65,535 bytes; in practice, however, the data link layer imposes a basic limit of 1,500 bytes on the payload of an ethernet frame (<xref ref-type="bibr" rid="B57">Schiavoni et al., 2013</xref>; <xref ref-type="bibr" rid="B39">IEEE, 2018</xref>).</p>
<p>With the headers for the data link (Ethernet), network (IPv4), and transport (UDP) layers accounted for, plus the audio header described above, in principle 1,452 bytes remain in each packet for audio data. Assuming 16-bit resolution, and the transmission of one UDP packet per audio buffer, data for up to 90 audio channels can be transmitted at a buffer size of 16 frames without fragmentation, or up to 45 channels at 32 frames.</p>
</sec>
</sec>
<sec id="s3-2">
<title>3.2 The networked audio client</title>
<p>Unlike the networked audio server, which runs on a general purpose computer and has access to threads of execution, which it can use to conduct related but separate tasks that rely on some central resource (the FIFO buffer alluded to above), the client implementation is designed to operate on a microcontroller platform that has no operating system, and no native notion of threads.<xref ref-type="fn" rid="fn26">
<sup>26</sup>
</xref>
</p>
<p>The task of the clients is threefold in nature:<list list-type="simple">
<list-item>
<p>1. To retrieve packets of audio data from the UDP multicast group;</p>
</list-item>
<list-item>
<p>2. To send a stream of audio data back to the multicast group, primarily to announce their connectivity;</p>
</list-item>
<list-item>
<p>3. To maintain, as far as possible, synchronous operation with the server, and (by extension) each other.</p>
</list-item>
</list>
</p>
<p>To address the first two requirements, the client sets up a socket, which it uses to both read from and write to the UDP multicast group.</p>
<p>The client was created as a C&#x2b;&#x2b; class named <monospace>NetJUCEClient</monospace>, an implementation of the Teensy Audio Library class <monospace>AudioStream</monospace>. <monospace>AudioStream</monospace> descendents must implement a method named <monospace>update()</monospace>; this method is called at each audio hardware interrupt, and is where an audio library class should perform operations on the current audio buffer. Networking operations are conducted from the method <monospace>NetJUCEClient::loop</monospace>. Avoiding conflicts with audio functionality, this method is called from Teensy&#x2019;s top level <monospace>loop()</monospace> function. A valid Teensy program must define a function by this name, and it is called repeatedly from the body of a non-terminating <monospace>while</monospace> loop throughout operation.</p>
<p>The two sets of operations are linked by way of an intermediate buffer, similar to the FIFO employed by the server. The client attempts to receive packets from, and, if it has generated a packet&#x2019;s worth of audio data, send a packet to, the multicast group on each call to <monospace>loop()</monospace>, with audio samples from incoming packets written to the intermediate buffer (see <xref ref-type="statement" rid="listing_3">Listing 3</xref>). The client also performs a periodic check for the presence of the server, and, as described in <xref ref-type="sec" rid="s3-2-1">Section 3.2.1</xref>, makes adjustments to its audio clock. When multiple clients are present, there are consequently multiple streams of audio packets reaching the multicast group. To avoid ambiguity and unnecessary packet reads at the client side, server and clients transmit audio data to the group on differing port numbers.</p>
<p>
<statement content-type="algorithm" id="listing_3">
<label>Listing 3. Loop method of the networked audio client implementation.</label>
<p>
<inline-graphic xlink:href="frvir-05-1391987-fx3.tif"/>
</p>
</statement>
</p>
<p>
<statement content-type="algorithm" id="listing_4">
<label>Listing 4. Update method of the networked audio client implementation.</label>
<p>
<inline-graphic xlink:href="frvir-05-1391987-fx4.tif"/>
</p>
<p>On each audio interrupt, the client reads from the intermediate buffer to produce samples for audio output. It also takes samples reaching its audio inputs and adds those to a packet to be sent to the multicast group at the earliest subsequent call to <monospace>NetJUCEClient::loop</monospace> (<xref ref-type="statement" rid="listing_4">Listing 4</xref>). The client&#x2019;s inputs can receive samples from any Teensy Audio Library object to which it has been connected programmatically; for round-trip time measurements the client&#x2019;s audio outputs were routed back to its inputs. An illustrative timeline of client-server interaction is depicted in <xref ref-type="fig" rid="F5">Figure 5</xref>.</p>
</statement>
</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Example timeline of interaction between the server and a client, via the UDP multicast group. Solid arrows indicate audio data being sent from the server to the multicast group; arrows with open heads indicate audio data being sent from the client back to the multicast group; arrows with dotted tails represent control data. Note that the server transmits to the multicast group irrespective of the presence of any client.</p>
</caption>
<graphic xlink:href="frvir-05-1391987-g005.tif"/>
</fig>
<sec id="s3-2-1">
<title>3.2.1 Synchronicity with the server</title>
<p>Due to the influence of clock drift and transmission jitter, and since the clients constitute a distributed system, with no direct knowledge of each other and no authoritative source of time, their third task posed the greatest challenge. A two-pronged strategy was developed for addressing server-client and inter-client timing discrepancies:</p>
<sec id="s3-2-1-1">
<title>3.2.1.1 Jitter compensation</title>
<p>Similar to the approach taken in prior work (<xref ref-type="bibr" rid="B55">Rushton et al., 2023</xref>), clients monitored their intermediate buffer for the difference between its write and read positions, using a delay-locked loop to keep this difference within an interval of one audio buffer&#x2019;s worth of frames. This was achieved by way of setting thresholds for the read-write difference, and adjusting the read-position increment if the difference fell beyond those thresholds; increasing the increment if the difference exceeded the high threshold; decreasing it should the difference fall short of the low threshold. This in turn entailed employing a fractional read-position, and interpolating around it to achieve an appropriate sample value; essentially a form of adaptive resampling. For this purpose a cubic Lagrange interpolator was used; sample values for the interpolator were converted from their 16-bit signed integer representation to floating point numbers, interpolation conducted, and the resulting value rounded to the nearest integer for output.</p>
</sec>
<sec id="s3-2-1-2">
<title>3.2.1.2 Clock drift compensation</title>
<p>In the absence of an authoritative source of time, clients were set up to infer the difference in rate between their own internal clock and that of the server by comparing the rate of packet reception from the network to their internal audio interrupt rate. This was achieved by taking the ratio, over thirty-second intervals, of packets written from the network to the intermediate buffer to blocks read from the intermediate buffer for audio output. This ratio was then used to calculate appropriate divisors to apply to the 24&#xa0;MHz master clock generated by a crystal oscillator on the Teensy, adjusting the audio clock&#x2019;s phase locked loop (PLL) to produce an adjusted audio sampling rate. The aim of this approach was to minimise reliance on the adaptive resampler described above, and ultimately encourage all clients to run at the same audio rate as the server.</p>
</sec>
</sec>
</sec>
<sec id="s3-3">
<title>3.3 The audio spatialisation algorithm</title>
<p>WFS was chosen for implementation due to the comparative ease with which the WFS algorithm can be parallelised. <xref ref-type="disp-formula" rid="e3">Equations 3</xref> and <xref ref-type="disp-formula" rid="e4">4</xref> (page 10) illustrate that the driving signal for a given secondary point source is dependent only on the signals and relative positions of the virtual primary sources, and is independent of the driving signals for the other secondary sources.</p>
<p>With some modifications, e.g., the possibility to specify speaker spacing parametrically, the WFS algorithm from (<xref ref-type="bibr" rid="B55">Rushton et al., 2023</xref>) was reused. This algorithm, facilitating the simulation of virtual primary sound sources, was written in Faust and compiled to a C&#x2b;&#x2b; class compatible with the Teensy Audio Library via Faust&#x2019;s <monospace>faust2teensy</monospace> utility. Hardware modules were connected to a general purpose computer via a USB hub and the <monospace>tycmd</monospace> utility from the TyTools suite (see <xref ref-type="sec" rid="s2-2">Section 2.2</xref>) was used to ensure that all modules were programmed with the same instructions.</p>
<p>As illustrated in <xref ref-type="fig" rid="F3">Figure 3</xref>, producing the driving signal for a WFS secondary source at position <inline-formula id="inf26">
<mml:math id="m30">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> entails applying a delay to an audio signal <inline-formula id="inf27">
<mml:math id="m31">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, representing the <inline-formula id="inf28">
<mml:math id="m32">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>th virtual primary source. This delay is based on the distance <inline-formula id="inf29">
<mml:math id="m33">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> between <inline-formula id="inf30">
<mml:math id="m34">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and the desired virtual position of <inline-formula id="inf31">
<mml:math id="m35">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. In its distributed form, the WFS algorithm, informed of the position in the array of the two loudspeakers for which it is responsible, computes only the delays for each primary source with respect to those two loudspeakers, i.e., for the <inline-formula id="inf32">
<mml:math id="m36">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>th virtual source, the <inline-formula id="inf33">
<mml:math id="m37">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>th hardware module computes <inline-formula id="inf34">
<mml:math id="m38">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf35">
<mml:math id="m39">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. To reduce the computational burden placed on the hardware modules, specifically with regard to memory, the length of the delay lines was reduced by discarding the longitudinal component of <inline-formula id="inf36">
<mml:math id="m40">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, leaving only the relative inter-speaker delay.</p>
<p>For the WFS prefilter, <inline-formula id="inf37">
<mml:math id="m41">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> was mapped to an inverse square law for frequency-independent amplitude loss to the virtual medium of propagation, and to the cutoff frequency of a two-pole lowpass filter defined by Faust&#x2019;s <monospace>fi.lowpass</monospace> function.<xref ref-type="fn" rid="fn27">
<sup>27</sup>
</xref> Adopting a modified version of <xref ref-type="disp-formula" rid="e4">Equation 4</xref>, the driving function becomes:<disp-formula id="e5">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<sec id="s3-3-1">
<title>3.3.1 Modularity and maximum delay</title>
<p>The reduction in the maximum delay length represented by the subtraction of the longitudinal distance component in <xref ref-type="disp-formula" rid="e5">Equation 5</xref> is essential for the viability of the system. As capable a platform as Teensy 4.1 is, as described in <xref ref-type="sec" rid="s2-2">Section 2.2</xref>, it is limited in terms of memory. This in turn places limits on the lengths of delay lines that it can compute, a matter exacerbated if there are many such delays to consider, such as in the case of a WFS implementation with numerous virtual sound sources. Each hardware module must compute two delay lines for each virtual source, one for each of its output channels, the maximum length of which (depending on the position of a given module in the speaker array) corresponds, after removal of the longitudinal component, to the width of the speaker array. It was observed that, for eight virtual sources and eight hardware modules, the maximum speaker spacing permissible lay at around .4&#xa0;m, corresponding with a speaker array of maximum width <inline-formula id="inf38">
<mml:math id="m43">
<mml:mrow>
<mml:mn>15</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>0.4</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 6&#xa0;m, equating to a maximum delay of <inline-formula id="inf39">
<mml:math id="m44">
<mml:mrow>
<mml:mo>&#x223c;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>17&#xa0;ms or approximately 795 samples at a sampling rate of 44.1&#xa0;kHz. The matter has not been rigorously tested, but nonetheless the presumption is that this places significant limits on the modularity of the system. Teensy&#x2019;s memory capacity can be extended by attaching up to two inexpensive PSRAM chips for a further 16&#xa0;MB of memory. These chips must be soldered onto the Teensy board, however, and the suitability of such additional memory for rapid access, such as is required in an audio DSP algorithm, remains to be investigated.</p>
</sec>
<sec id="s3-3-2">
<title>3.3.2 Controlling the WFS algorithm</title>
<p>Parameter values are delivered to the Faust algorithm in the form of Open Sound Control (OSC) messages. OSC control data, describing virtual sound source positions, speaker spacing, and informing clients of their position in the speaker array, is bundled into UDP packets and delivered by the server to the multicast group for all clients to consume. Source positions are described as coordinates in a two-dimensional plane, with <inline-formula id="inf40">
<mml:math id="m45">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf41">
<mml:math id="m46">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> components, each normalised to the range [0,1], transmitted separately. The width of the speaker array is inferred from the speaker spacing (in metres), multiplied by the number of speaker-intervals in the array, i.e., one fewer than the number of speakers. For the proposed implementation, the number of speakers is known to the server and clients at compile time; with further development this could be made specifiable at runtime. Similarly, at the time of writing, the longitudinal depth of the virtual sound field is hard-coded into the clients and will be generalised in a future iteration of the system.</p>
<p>
<xref ref-type="statement" rid="listing_5">Listing 5</xref> demonstrates an example control data packet, an OSC bundle containing one message. This message has address<monospace>/source/0/x</monospace>, indicating that it refers to the <inline-formula id="inf42">
<mml:math id="m47">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-coordinate of the zeroth sound source, providing a value in the form of a big-endian 32-bit floating point number, <monospace>0x3d1b5fa2</monospace>, approximately 0.038<sub>10</sub>.</p>
<p>
<statement content-type="algorithm" id="listing_5">
<label>Listing 5. Network capture: ethernet frame containing a UDP control data packet.</label>
<p>
<inline-graphic xlink:href="frvir-05-1391987-fx5.tif"/>
</p>
</statement>
</p>
</sec>
</sec>
<sec id="s3-4">
<title>3.4 System overview</title>
<sec id="s3-4-1">
<title>3.4.1 Hardware setup</title>
<p>The networked audio server runs on a general purpose computer. Throughout development, testing and evaluation, that computer was an ASUS G513R Notebook PC, with an AMD Ryzen 7 6800H processor with a clock speed of 3.2&#xa0;GHz. For the majority of development, the computer&#x2019;s internal sound card was used; for testing and evaluation, it was connected to a Steinberg UR44C USB audio interface, the hope being that external hardware would provide more consistent audio interrupt timing, thus minimising jitter originating at the server.</p>
<p>The computer was connected via CAT6 ethernet cable to an eight-port ethernet switch (D-Link DGS-108GL). For evaluation, and to support a total of eight networked audio clients, this switch was daisy-chained to an additional switch (D-Link DES-1008D). Teensy 4.1 hardware modules, assembled as per <xref ref-type="fig" rid="F6">Figure 6</xref>, were connected via CAT6 ethernet cables to available ports on the ethernet switches. Hardware modules were powered by a combination of a seven-port USB hub, plus, for the eighth module, a USB mains socket. The two audio outputs of each hardware module were connected to M-Audio BX5 loudspeakers.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>A hardware module consisting of Teensy 4.1 microcontroller (labelled with the last 2&#xa0;bytes of its serial number-derived IP address), connected via headers to an audio shield and via ribbon cable to an ethernet shield.</p>
</caption>
<graphic xlink:href="frvir-05-1391987-g006.tif"/>
</fig>
</sec>
<sec id="s3-4-2">
<title>3.4.2 Software system</title>
<p>Server-side, the software system consists of a VST plugin running in Reaper digital audio workstation software.<xref ref-type="fn" rid="fn28">
<sup>28</sup>
</xref> The plugin comprises the networked audio server, receiving monophonic audio sources in the form of audio or instrument tracks in the DAW, plus a control data server, commanded either by parameter automation via the DAW, or manually via a graphical user interface (see <xref ref-type="fig" rid="F7">Figure 7</xref>). The audio and control data servers send streams of UDP packets to a UDP multicast group.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>User interface for the WFS controller DAW plugin, with modal settings window visible. The interface consists of an X/Y control surface, with eight nodes representing the coordinates, normalised to <inline-formula id="inf43">
<mml:math id="m48">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mn>0,1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, of sound sources in a virtual sound field. Dropdown menus at the bottom of the interface correspond with hardware module positions in the loudspeaker array; there are eight such menus in total, each associated hardware module producing output for two loudspeakers. The settings window facilitates specifying the speaker spacing, and shows a list of connected network peers.</p>
</caption>
<graphic xlink:href="frvir-05-1391987-g007.tif"/>
</fig>
<p>Client-side software connects to the multicast group and reads UDP packets containing audio and control data from the server. These streams are delivered to the Faust-based WFS algorithm, with audio streams processed according to the control parameters of virtual sound source positions and speaker spacing. The WFS algorithm produces driving signals for each of the two output channels of the hardware module on which it is running. Additionally, the client-side networked audio client returns a stream of audio data to the multicast group, to be consumed by the server.</p>
<p>Code for the server and client software components can be found at <ext-link ext-link-type="uri" xlink:href="https://github.com/hatchjaw/netjuce">https://github.com/hatchjaw/netjuce</ext-link> and <ext-link ext-link-type="uri" xlink:href="https://github.com/hatchjaw/netjuce-teensy">https://github.com/hatchjaw/netjuce-teensy</ext-link> respectively.</p>
</sec>
</sec>
</sec>
<sec sec-type="results" id="s4">
<title>4 Results</title>
<p>Possessing technical underpinnings, but ultimately being designed to serve immersive auditory ends, it was important to consider the performance of the system described and developed in <xref ref-type="sec" rid="s3">Section 3</xref> in terms of both its technical capabilities and the quality of the perceptual effects it was able to support. The success of the system as a platform for audio spatialisation techniques is contingent on it being composed of effective solutions to the challenges posed by distributing audio processing across a local area network. It is of limited worth, however, as a technical exercise in isolation; the subjective assessment of listeners may help identify the most critical aspects of the technical implementation and guide future development.</p>
<sec id="s4-1">
<title>4.1 Technical evaluation</title>
<p>Of most pressing technical concern is the matter of synchronicity between the hardware modules. To assess this, a similar approach was taken to that found in (<xref ref-type="bibr" rid="B55">Rushton et al., 2023</xref>; <xref ref-type="bibr" rid="B32">Gabrielli et al., 2012</xref>).</p>
<sec id="s4-1-1">
<title>4.1.1 Round trip time</title>
<p>To measure transmission round trip time (RTT), the server transmitted a unipolar sawtooth wave of unit amplitude increment to the multicast group, and each client, upon receiving the signal simply returned it immediately to the group to be read by the server. At the server side, the return signal, <inline-formula id="inf44">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, was subtracted from the outgoing signal, <inline-formula id="inf45">
<mml:math id="m50">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, at the time of reception, with round trip time found as:<disp-formula id="e6">
<mml:math id="m51">
<mml:mrow>
<mml:mtext>RTT</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="cases">
<mml:mtr>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mn>16</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mtext>otherwise</mml:mtext>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>where <inline-formula id="inf46">
<mml:math id="m52">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mn>16</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the maximum value representable by a signed 16-bit integer, <monospace>0x7fff</monospace> (32 767<sub>10</sub>).</p>
<p>The resulting value is the number of samples elapsed between transmission and reception (see <xref ref-type="fig" rid="F8">Figure 8</xref>). Since there is one source of transmission, for multiple clients, comparing RTT offers a means to assess inter-client synchronicity. Server-to-client latency cannot be measured in this way, but that can be inferred to be around half of, and, of course, certainly not greater than, the RTT.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Illustration of the use of a test signal, a unipolar sawtooth wave, to measure round trip time. Subtracting the return signal from the outgoing signal gives the time (in samples) between transmission and reception.</p>
</caption>
<graphic xlink:href="frvir-05-1391987-g008.tif"/>
</fig>
</sec>
<sec id="s4-1-2">
<title>4.1.2 Clock drift/skew</title>
<p>A unipolar sawtooth wave of unit amplitude increment was generated on the clients, subtracted from the incoming sawtooth wave from the server, and the difference (found as per <xref ref-type="disp-formula" rid="e6">Equation 6</xref>) returned to the multicast group for consumption by the server. The incoming signal and the one being generated on a given client should, under ideal conditions, be out of phase by some constant value; if this value changes then relative drift has occurred between server and client. The client-side clock-adjustment strategy was designed to minimise the reliance on the adaptive resampling approach that it complements; low drift would be indicative of the effectiveness of that strategy.</p>
<p>Initial RTT and relative drift measurements for eight clients are shown in <xref ref-type="fig" rid="F9">Figure 9</xref>. Mean RTT spread, describing the average temporal interval over which clients were distributed over the course of the test, is promising, the 12.43 sample interval corresponding with approximately 282&#xa0;&#xb5;s. RTT is clustered around a respectable 190 samples (<inline-formula id="inf47">
<mml:math id="m53">
<mml:mrow>
<mml:mo>&#x223c;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>4.3&#xa0;&#xb5;s).</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Round-trip time, RTT spread, and clock drift measurements for eight networked audio clients, for a networked audio session of 8&#xa0;minutes&#x2019; duration. Audio buffer size, 16 frames, sampling rate 44.1&#xa0;kHz. The legend in the bottommost plot applies also to the upper plot. Round-trip time, in samples, measured at the server, and found as the difference between an outgoing sawtooth wave and its returning counterparts from each of eight connected clients. Round-trip time spread found as the range (<inline-formula id="inf48">
<mml:math id="m54">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>RTT</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mtext>RTT</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>), in samples, at each point in time, of round-trip times reported for all eight clients. Mean spread is the arithmetic mean of RTT spread values taken across the entire test. Drift, in samples, found as the difference between an outgoing sawtooth wave and a sawtooth wave generated on each client, that difference returned to the multicast group for consumption by the server.</p>
</caption>
<graphic xlink:href="frvir-05-1391987-g009.tif"/>
</fig>
<p>That visual clustering, coupled with the apparent tendency for RTT spread to lie at around 16 samples (i.e., precisely one buffer), suggest, however, a certain over-aggressiveness in the resampling strategy, perhaps resulting in a polarisation of clients to the temporal extremes of the interval between their audio interrupts. What <xref ref-type="fig" rid="F9">Figure 9</xref> does not show, and, given the short timescales involved, is not easily represented in such a diagram, is the rate of relative inter-client movement, i.e., the rate of change of asynchronicity. Subjective assessment of the system&#x2019;s audible output revealed that, given the rapid rate of relative movement between clients, in this state it would not stand up to perceptual testing.</p>
<p>Transmitting a white Gaussian noise signal to the clients and delivering this to their audio outputs without further processing&#x2014;seeking, essentially, to sonify QoS&#x2014;an aggressive phasing, or time-varying comb-filter effect was clearly audible. This effect is visualised in <xref ref-type="fig" rid="F10">Figure 10A</xref>; ideally (subject to the frequency response of the microphone used) an ambient recording of a white noise source would correspond with a magnitude spectrogram exhibiting equal intensity across the frequency range at all times; clearly, though, there are regions of greater and lesser intensity, and these regions shift and change rapidly over time. In addition to the above, tests involving the reproduction of signals containing steady-state harmonic content revealed obtrusive audible artefacts.</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>Magnitude spectrograms of ambient, monophonic recordings of a reproduction of white Gaussian noise by a group of eight networked audio clients driving an array of fifteen loudspeakers spaced at intervals of .175&#xa0;m. Capacitor microphone placed <inline-formula id="inf49">
<mml:math id="m55">
<mml:mrow>
<mml:mo>&#x223c;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 2&#xa0;m from the speaker array. Audio buffer size <bold>(A)</bold> 16 frames; <bold>(B)</bold> 32 frames.</p>
</caption>
<graphic xlink:href="frvir-05-1391987-g010.tif"/>
</fig>
<p>A buffer size of 16 frames had been selected in an attempt to minimise the duration of the window of inter-client synchronicity, and to maximise the number of channels that could be transmitted over the network, subject to restrictions posed by the MTU (see <xref ref-type="sec" rid="s3-1-3">Section 3.1.3</xref>). Recalling, however, that previous work (<xref ref-type="bibr" rid="B55">Rushton et al., 2023</xref>) had employed a 32-frame audio buffer, equivalent measurements were taken for the larger buffer size, the results of which are depicted in <xref ref-type="fig" rid="F10">Figures 10B</xref>, <xref ref-type="fig" rid="F11">11</xref>.</p>
<fig id="F11" position="float">
<label>FIGURE 11</label>
<caption>
<p>Round-trip time, RTT spread, and clock drift measurements for eight networked audio clients, for a networked audio session of 8&#xa0;minutes&#x2019; duration. Audio buffer size, 32 frames.</p>
</caption>
<graphic xlink:href="frvir-05-1391987-g011.tif"/>
</fig>
<p>Again, visually, there is an apparent clustering in the RTT recordings, with clients spending large periods separated by around one buffer&#x2019;s worth of samples (<inline-formula id="inf50">
<mml:math id="m56">
<mml:mrow>
<mml:mo>&#x223c;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>726&#xa0;&#xb5;s), seemingly often grouped at either extreme of the interval of one audio buffer. The mean RTT spread, equating to <inline-formula id="inf51">
<mml:math id="m57">
<mml:mrow>
<mml:mo>&#x223c;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>626 &#xb5;s, is comparable with results from prior work. Importantly, however, and as demonstrated in <xref ref-type="fig" rid="F10">Figure 10B</xref>, the rate of relative inter-client temporal movement was much improved by the switch to a 32-frame buffer. Although exhibiting similar visual striations to the spectrogram for the test at 16 frames, fluctuations occur less frequently, and seemingly more gradually. Indeed, subjectively-speaking, the disruption caused by the phasing effect that afflicted the 16-frame buffer implementation was significantly reduced, as was the presence of audible artefacts affecting harmonic signals. Thus it was the version of the system employing a buffer size of 32 frames that was exposed to perceptual evaluation.</p>
<p>Clock drift measurements in <xref ref-type="fig" rid="F9">Figures 9</xref>, <xref ref-type="fig" rid="F11">11</xref> exhibit comparable trends. Increasing negative drift over time is indicative of the clients running faster than the server. Visually, there is evidence that client clocks adjust to approximate parity with the server for periods of time, perhaps falling slightly slower (e.g., the drift plot in <xref ref-type="fig" rid="F9">Figure 9</xref>, between 90 and 160&#xa0;s), but intermittently demonstrate leaps, the largest of which are in the negative direction.</p>
</sec>
<sec id="s4-1-3">
<title>4.1.3 Discussion</title>
<p>The temporal clustering and polarisation seen in <xref ref-type="fig" rid="F9">Figures 9</xref>, <xref ref-type="fig" rid="F11">11</xref> is indicative of two points for improvement with regard to technical implementation: the read-write difference threshold strategy may be insufficiently forgiving, forcing the read position into rapid changes in response to periods of jitter; and without an authoritative clock to indicate to the clients when each block of audio data should be output, even if clock rates were perfectly aligned, there is nothing to guarantee agreement of the timing of audio interrupts at the client side.</p>
<p>The steps seen in the drift plots in <xref ref-type="fig" rid="F9">Figures 9</xref>, <xref ref-type="fig" rid="F11">11</xref> appear (particularly in <xref ref-type="fig" rid="F11">Figure 11</xref>) to occur in multiples of the audio buffer size. This suggests either packet loss or, more likely,<xref ref-type="fn" rid="fn29">
<sup>29</sup>
</xref> moments of pronounced jitter, causing clients to rapidly reduce (or, less commonly, increase) their read-position increment to maintain the read-write delta. One may expect such a phenomenon to be followed by an immediate rebound, but this appears to be a more gradual process. The overall trend, for thiscombination of server and clients at least, is for clients to run faster than the server; the clock adjustment strategy employed relies on inferring time from the rate of packet transmission, which may not offer sufficient temporal resolution for accurate drift compensation.</p>
</sec>
</sec>
<sec id="s4-2">
<title>4.2 Perceptual evaluation</title>
<p>The WFS system was subjected to an informal perceptual evaluation, a localisation experiment of a similar form to that presented by Verheijen (<xref ref-type="bibr" rid="B63">Verheijen, 1998</xref>, ch. 6), albeit with the inclusion of simulated distance as well as lateral position. Participants were presented with a virtual sound source at various locations and asked to indicate, on a diagram of the virtual sound field, the point from which they estimated the sound had emanated. The informality of the experiment arose in part as a consequence of the listening environment not being acoustically treated, and there being sources of ambient sound in the laboratory in which the WFS system was installed. Furthermore, the speaker array (<xref ref-type="fig" rid="F12">Figure 12</xref>) consisting of fifteen speakers, but each hardware module producing two audio output channels, the second channel of the right-most module was not used; for eight modules, however, the WFS plugin assumed a virtual sound field spanning sixteen speakers, thus it was possible to position a virtual sound source horizontally beyond the rightmost extent of the speaker array. Ultimately the aim of the experiment was to draw some preliminary, guiding conclusions as to the effectiveness of the distributed WFS system in triggering listeners&#x2019; localisation cues, its technical and installation shortcomings notwithstanding.</p>
<fig id="F12" position="float">
<label>FIGURE 12</label>
<caption>
<p>System configuration for technical and perceptual evaluation. Eight hardware modules connected to fifteen loudspeakers&#x2014;seven of the modules produced output for two loudspeakers each; the final module used only its first output channel.</p>
</caption>
<graphic xlink:href="frvir-05-1391987-g012.tif"/>
</fig>
<p>It was felt that listeners would be most comfortable localising a naturalistic sound, so, rather than use bursts of white noise as in (<xref ref-type="bibr" rid="B63">Verheijen, 1998</xref>), and wishing to minimise the potential effects of frequency-dependent localisation interference due to spatial aliasing, a broadband stimulus was selected in the form of a close-mic recording of a snare drum. The recorded sample was repeated three times in succession at intervals of .125&#xa0;s, and, again in the interests of adding a natural quality to the sound, with slight variations in amplitude (the second iteration of the sample was played marginally quieter than the first; the third slightly louder).</p>
<p>The system was presented to eight participants; a mixture of masters and PhD students aged between 22 and 38, with knowledge of audio and interactive computer systems. Participants were given a brief description of the system under evaluation, and informed that they should expect to hear sounds that appeared to emanate from &#x2018;behind&#x2019; the speaker array, from which they stood at a distance of <inline-formula id="inf52">
<mml:math id="m58">
<mml:mrow>
<mml:mo>&#x223c;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>2&#xa0;m. Eight different virtual source positions were specified via automation of the <inline-formula id="inf53">
<mml:math id="m59">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (lateral) and <inline-formula id="inf54">
<mml:math id="m60">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (longitudinal distance) components of the position of a node in the WFS plugin interface. The range of the <inline-formula id="inf55">
<mml:math id="m61">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> component corresponded with the distance from the centre of the driver of the leftmost speaker to that of the missing 16th loudspeaker; drivers lay at intervals of .175&#xa0;m, giving a horizontal axis spanning 2.625&#xa0;m. Longitudinal position was mapped to a range from 0&#xa0;m (i.e., lying directly on the speaker array) to 10&#xa0;m &#x201c;behind&#x201d; the array.</p>
<p>For each position, the auditory stimulus was played, and repeated at the participant&#x2019;s request. Each participation was five to 10&#xa0;minutes in duration. Details of the source positions for each test, and responses for the eight participants, are displayed in <xref ref-type="fig" rid="F13">Figure 13</xref>.</p>
<fig id="F13" position="float">
<label>FIGURE 13</label>
<caption>
<p>Results of a localisation experiment based on WFS virtual primary sources produced by the proposed system. Lateral (horizontal axis) and longitudinal (vertical axis) components are normalised to <inline-formula id="inf56">
<mml:math id="m62">
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mn>0,1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. Each plot represents the virtual sound field; the horizontal axis (i.e., longitudinal component equalling 0) corresponds with the location of the speaker array. Participants stood at a distance of approximately 2&#xa0;m from the array. Each plot shows the intended position of the virtual sound source as specified by parameters to the WFS plugin interface (cross) and estimated sound source positions as reported by participants (dots). Each plot is labelled with the mean Euclidean error <inline-formula id="inf57">
<mml:math id="m63">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> between intended position and reported positions. Legend in plot <bold>(A)</bold> also applies to plots <bold>(B)</bold> to <bold>(H)</bold>.</p>
</caption>
<graphic xlink:href="frvir-05-1391987-g013.tif"/>
</fig>
<p>As can be seen, although demonstrating significant outliers (e.g., the position reported by the fifth participant for test <bold>(D)</bold>), certain trends do appear to emerge from the results. Firstly, responses loosely track the intended positions, with reported positions most closely corresponding with intended ones for virtual source locations lying close to the speaker array. Indeed, tests <bold>(B)</bold>, <bold>(D)</bold>, and <bold>(G)</bold> exhibit the lowest mean error values between the intended and reported positions. The results for tests <bold>(C)</bold> and <bold>(H)</bold>, exhibit the greatest mean error, and ambiguity regarding the lateral position of distant sound sources is perhaps to be expected; as the distance of a sound source from the listener increases, <inline-formula id="inf58">
<mml:math id="m64">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> tends towards zero, and thus the ITD (and ILD) also approaches zero; thus, with increased distance the wavefront produced by a sound source (be it a real sound source or one synthesised under ideal conditions) approximates more and more closely a plane wave. In any case, despite this inherent, physical ambiguity, there is a tendency in results <bold>(C)</bold> and <bold>(H)</bold> toward the lateral location of the most longitudinally distant intended virtual source positions. Particularly for test <bold>(C)</bold>, participants seem to have had greater difficulty in estimating the depth of the virtual sound field; this may simply be as a function of their developing a familiarity with that aspect of it over the course of what was only a brief experiment.</p>
<p>Participants were asked for any anecdotal observations they had, based on their experience of the experiment. One participant noted, for the first position in particular, that the amplitude variations between the snare drum strikes gave the impression of a sound source that was advancing upon the listening position; for the lack of any visual cue as to the position of the sound source, this is a reasonable conclusion to draw; it did not, however, ultimately prevent them from reaching a decision with regard to their estimate for the position of the sound source. Another, likely hearing the time-varying comb-filter effect, asked whether the &#x201c;phasing&#x201d; they were hearing was intentional. A third, also perceiving a similar phenomenon, suggested that they felt that the sound sources were moving. Finally, a participant with prior experience working with WFS systems, remarked that the distance effect (i.e., the WFS prefilter) was perhaps a little extreme, and not altogether realistic.</p>
<sec id="s4-2-1">
<title>4.2.1 Discussion</title>
<p>The phasing effect noted by one participant is a consequence of the approach taken to combating jitter and keeping the clients close, temporally, together, and as close to the server as possible. The current approach is likely too aggressive to be viable for high-quality audio output across a broad array of source signals. A transient, unpitched sound source like a snare drum, though perhaps audibly susceptible to the time-varying comb-filter effect described, masks other artefacts caused by phenomena such as rapid fluctuations in the clients&#x2019; buffer read position increment, and sudden, comparatively large audio clock adjustments.</p>
<p>The above being said, the perceptual test indicates that the system produces virtual sound sources that listeners are, at least to some extent, able to localise. Further, it achieves this at a significantly lower cost-per-channel than any of the systems discussed in sections <xref ref-type="sec" rid="s2-3-3">Section 2.3.3</xref>, and, speakers and cables excepted, compares favourably with the OTTOSonics system referred to in <xref ref-type="sec" rid="s2-4">Section 2.4</xref>, particularly as channel-count increases, i.e., in terms of cost, it has the potential to scale better. The most costly component of the system is the computer, but this could be exchanged for any interested user&#x2019;s personal machine, so long as it is able to run a DAW and has an ethernet interface. The Teensy modules, including audio shield, cost around &#x20ac;45; eight-port ethernet switches can be purchased for as little as &#x20ac;20&#x2013;30. Assuming a computer costing &#x20ac;1,500, at 16 channels we can estimate around &#x20ac;120/channel, dropping to &#x20ac;50/channel for 64 channels.</p>
</sec>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>In this article we have described the development of a novel, distributed system for audio spatialisation. The proposed networked audio system, featuring low-cost, microcontroller-based clients, represents a milestone on the road towards an accessible alternative to state of the art spatial and immersive audio installations. Evaluation of the system exposed the extent of the technical challenges that confront it in its current form, but revealed that it may offer performance sufficient to support timing-critical sound field synthesis techniques.</p>
<p>Client asynchronicity may affect the integrity of the spatial audio algorithm and give rise to audible artefacts, and the strategies presented here for mitigation of asynchronicity call for further refinement. Ultimately, an authoritative source of time should be sought, either in the form of a shared physical clock or a PTP implementation; in light of the disruptive potential of the system, care should be taken, however, to find a solution that is cost-effective and easy to replicate.</p>
<p>Plans for future research include an exploration of other potential hardware platforms for the network client. A successor to Teensy 4.1 may provide more memory or support higher-quality audio, for example. Only briefly considered here, the Raspberry Pi family of embedded Linux systems is priced comparably with the Teensy and supports audio breakout boards capable of 24-bit audio, with the added benefit of significantly greater memory. Further, the Raspberry Pi Compute Module 4 is capable of physical layer timestamping and thus may facilitate the creation of a system synchronised via hardware PTP.</p>
<p>A basic, linear, wave field synthesis algorithm has been demonstrated, implementing virtual primary sound sources; the client implementation and control software should be generalised to support nonlinear speaker arrays, plane and focused sources, and better models for energy absorption. A self-calibrating system, whereby clients <italic>discover</italic> their position in an installation rather than needing to be informed of it, would represent an additional boost to accessibility. Higher order ambisonics, subject to an assessment of its suitability to parallelisation, remains as a worthy target for implementation in future work. Further, convolution-heavy DSP algorithms, such as those supporting virtual acoustics and auralization, may be well-served by the extent of the computational resources afforded by our distributed system.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec sec-type="ethics-statement" id="s7">
<title>Ethics statement</title>
<p>Ethical approval was not required for the studies involving humans because participants gave their age, but no other identifying information, and their participation was not filmed or otherwise recorded. The studies were conducted in accordance with the local legislation and institutional requirements. The participants provided their written informed consent to participate in this study.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>TR: Conceptualization, Investigation, Methodology, Software, Visualization, Writing&#x2013;original draft, Writing&#x2013;review and editing. RM: Conceptualization, Resources, Supervision, Writing&#x2013;review and editing. SS: Resources, Supervision, Writing&#x2013;review and editing. TR: Resources, Supervision, Writing&#x2013;review and editing. SL: Supervision, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received forthe research, authorship, and/or publication of this article. This project was funded by the FAST ANR project (ANR-20-CE38-0001) and the &#x201c;moyens incitatifs&#x201d; program of the Lyon Inria center.</p>
</sec>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The author(s) declared that they were an editorial board member of Frontiers, at the time of submission. This had no impact on the peer review process and the final decision.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn id="fn1">
<label>1</label>
<p>Consumer-grade, eight-port ethernet switches can cost as little as &#x20ac;20; The cheapest equivalent devices with PTP support cost, at the time of writing, on the order of &#x20ac;150&#x2013;200, e.g., <ext-link ext-link-type="uri" xlink:href="https://www.fs.com/de-en/products/148180.html">https://www.fs.com/de-en/products/148180.html</ext-link> &#x2014; All URLs verified 12/01/2024.</p>
</fn>
<fn id="fn2">
<label>2</label>
<p>See, for example, <ext-link ext-link-type="uri" xlink:href="https://github.com/tschiemer/aes67">https://github.com/tschiemer/aes67</ext-link> and <ext-link ext-link-type="uri" xlink:href="https://github.com/adiknoth/Open-AVB">https://github.com/adiknoth/Open-AVB</ext-link>
</p>
</fn>
<fn id="fn3">
<label>3</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://arduino.cc/">https://arduino.cc/</ext-link>
</p>
</fn>
<fn id="fn4">
<label>4</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://github.com/rsta2/circle">https://github.com/rsta2/circle</ext-link>
</p>
</fn>
<fn id="fn5">
<label>5</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://faust.grame.fr/">https://faust.grame.fr/</ext-link>
</p>
</fn>
<fn id="fn6">
<label>6</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://faustdoc.grame.fr/manual/tools/">https://faustdoc.grame.fr/manual/tools/</ext-link>
</p>
</fn>
<fn id="fn7">
<label>7</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://pjrc.com/store/teensy41.html">https://pjrc.com/store/teensy41.html</ext-link>
</p>
</fn>
<fn id="fn8">
<label>8</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://electro-smith.com/daisy/daisy">https://electro-smith.com/daisy/daisy</ext-link>
</p>
</fn>
<fn id="fn9">
<label>9</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://espressif.com/en/products/devkits/esp-audio-devkits">https://espressif.com/en/products/devkits/esp-audio-devkits</ext-link>
</p>
</fn>
<fn id="fn10">
<label>10</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://st.com/en/evaluation-tools/stm32h747i-disco.html">https://st.com/en/evaluation-tools/stm32h747i-disco.html</ext-link>
</p>
</fn>
<fn id="fn11">
<label>11</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://shop.bela.io/products/bela-starter-kit">https://shop.bela.io/products/bela-starter-kit</ext-link>
</p>
</fn>
<fn id="fn12">
<label>12</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://beagleboard.org/black">https://beagleboard.org/black</ext-link>
</p>
</fn>
<fn id="fn13">
<label>13</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://www.raspberrypi.com/products/raspberry-pi-4-model-b/">https://www.raspberrypi.com/products/raspberry-pi-4-model-b/</ext-link>
</p>
</fn>
<fn id="fn14">
<label>14</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://koromix.dev/tytools">https://koromix.dev/tytools</ext-link>
</p>
</fn>
<fn id="fn15">
<label>15</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://github.com/GameOfLife/WFSCollider">https://github.com/GameOfLife/WFSCollider</ext-link>
</p>
</fn>
<fn id="fn16">
<label>16</label>
<p>See also <ext-link ext-link-type="uri" xlink:href="https://tu.berlin/en/ak/research/projects/wellenfeldsynthese-fuer-einen-grossen-hoersaal">https://tu.berlin/en/ak/research/projects/wellenfeldsynthese-fuer-einen-grossen-hoersaal</ext-link> and WFS speaker module produced by Four Audio for installation at TU <ext-link ext-link-type="uri" xlink:href="https://four-audio.com/en/products/wfs/">https://four-audio.com/en/products/wfs/</ext-link>
</p>
</fn>
<fn id="fn17">
<label>17</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://www.ircam.fr/article/connaissez-vous-lespace-de-projection">https://www.ircam.fr/article/connaissez-vous-lespace-de-projection</ext-link>
</p>
</fn>
<fn id="fn18">
<label>18</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://empac.rpi.edu/about/building/venues">https://empac.rpi.edu/about/building/venues</ext-link>
</p>
</fn>
<fn id="fn19">
<label>19</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://holoplot.com/insights/case-studies/msg-sphere-case-study">https://holoplot.com/insights/case-studies/msg-sphere-case-study</ext-link>
</p>
</fn>
<fn id="fn20">
<label>20</label>
<p>A successor to the defunct CoreAudio/JACK bridge has been proposed but remains unrealised: <ext-link ext-link-type="uri" xlink:href="https://github.com/jackaudio/jack-router/blob/main/macOS/docs/JackRouter-AudioServerPlugin.md">https://github.com/jackaudio/jack-router/blob/main/macOS/docs/JackRouter-AudioServerPlugin.md</ext-link>. This issue of course also affects the viability of the JackTrip-based approach.</p>
</fn>
<fn id="fn21">
<label>21</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://github.com/jacktrip/jacktrip/blob/v1.6.8/src/AudioInterface.h\#L56">https://github.com/jacktrip/jacktrip/blob/v1.6.8/src/AudioInterface.h\&#x23;L56</ext-link>
</p>
</fn>
<fn id="fn22">
<label>22</label>
<p>Often used interchangeably with the word <italic>sample</italic>, a <italic>frame</italic> represents the samples for all channels for a given sample instant; thus the number of frames in a network packet or audio buffer is the number of samples divided by the number of channels.</p>
</fn>
<fn id="fn23">
<label>23</label>
<p>JUCE 7.0.5 <ext-link ext-link-type="uri" xlink:href="https://github.com/juce-framework/JUCE">https://github.com/juce-framework/JUCE</ext-link>
</p>
</fn>
<fn id="fn24">
<label>24</label>
<p>Endianness is a thorny issue&#x2014;just consider Danny Cohen&#x2019;s <italic>&#x2026;Plea for Peace</italic> (<xref ref-type="bibr" rid="B21">Cohen, 1981</xref>). To appeal momentarily to authority, JackTrip too transmits audio data (and port numbers, etc.) little-endian, &#x201c;network byte order&#x201d; notwithstanding.</p>
</fn>
<fn id="fn25">
<label>25</label>
<p>Subscript 10 is employed here to indicate a decimal number.</p>
</fn>
<fn id="fn26">
<label>26</label>
<p>There is in fact a non-core library, <italic>TeensyThreads</italic>, that provides thread-like functionality. It was experimented with during development, but found to be incompatible with the interrupt-driven nature of the Teensy audio and networking libraries.</p>
</fn>
<fn id="fn27">
<label>27</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://faustlibraries.grame.fr/libs/filters/\#filowpass">https://faustlibraries.grame.fr/libs/filters/\&#x23;filowpass</ext-link>
</p>
</fn>
<fn id="fn28">
<label>28</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://reaper.fm/">https://reaper.fm/</ext-link>
</p>
</fn>
<fn id="fn29">
<label>29</label>
<p>During many hours of testing, no instance of packet loss was reported by any of the clients.</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Adriaensen</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2005</year>). &#x201c;<article-title>Using a DLL to filter time</article-title>,&#x201d; in <source>
<italic>Linux audio conference</italic> (Karlsruhe, Germany)</source>.</citation>
</ref>
<ref id="B2">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Adriaensen</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2012</year>). &#x201c;<article-title>Controlling adaptive resampling</article-title>,&#x201d; in <conf-name>10th International Linux Audio Conference Stanford, CA, USA: CCRMA</conf-name> (<publisher-name>Stanford University</publisher-name>), <fpage>145</fpage>&#x2013;<lpage>151</lpage>.</citation>
</ref>
<ref id="B3">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ahrens</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2012</year>). <source>
<italic>Analytic Methods of sound field synthesis</italic>. T-labs series in telecommunication services</source>. <publisher-name>Springer Science and Business Media</publisher-name>.</citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ahrens</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Rabenstein</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Spors</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2008</year>). <source>The theory of wave field synthesis revisited</source>. <publisher-loc>Amsterdam, Netherlands</publisher-loc>: <publisher-name>Audio Engineering Society</publisher-name>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>AL-Dhief</surname>
<given-names>F. T.</given-names>
</name>
<name>
<surname>Sabri</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Latiff</surname>
<given-names>N. A.</given-names>
</name>
<name>
<surname>Malik</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Abbas</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Albader</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Performance comparison between TCP and UDP protocols in different simulation scenarios</article-title>. <source>Int. J. Eng. and Technol.</source> <volume>7</volume>, <fpage>172</fpage>&#x2013;<lpage>176</lpage>. <pub-id pub-id-type="doi">10.14419/ijet.v7i4.36.23739</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Baalman</surname>
<given-names>M. A. J.</given-names>
</name>
<name>
<surname>Hohn</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Schampijer</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Koch</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2007</year>). &#x201c;<article-title>Renewed architecture of the sWONDER software for Wave Field Synthesis on large scale systems</article-title>,&#x201d; in <source>
<italic>Proceedings of the 5th int. Linux audio conference</italic> (Berlin, Germany)</source>.</citation>
</ref>
<ref id="B7">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Bakker</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Cooper</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kitagawa</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2014</year>). <source>An introduction to networked audio</source>. <publisher-loc>Rellingen, Germany</publisher-loc>: <publisher-name>White Paper, Yamaha Commercial Audio Team</publisher-name>.</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Belloch</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Bad&#xed;a</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Larios</surname>
<given-names>D. F.</given-names>
</name>
<name>
<surname>Personal</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Ferrer</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Fuster</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>On the performance of a GPU-based SoC in a distributed spatial audio system</article-title>. <source>J. Supercomput.</source> <volume>77</volume>, <fpage>6920</fpage>&#x2013;<lpage>6935</lpage>. <pub-id pub-id-type="doi">10.1007/s11227-020-03577-4</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Berger</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Farzaneh</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Murakami</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Valentin</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Exploring the past with virtual acoustics and virtual reality</article-title>,&#x201d; in <source>
<italic>2023 Immersive and 3D audio: from Architecture to automotive</italic> (bologna, Italy: IEEE)</source>.</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Berkhout</surname>
<given-names>A. J.</given-names>
</name>
</person-group> (<year>1988</year>). <article-title>A holographic approach to acoustic control</article-title>. <source>J. Audio Eng. Soc.</source> <volume>36</volume>, <fpage>977</fpage>&#x2013;<lpage>995</lpage>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Berkhout</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>de Vries</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Vogel</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>1993</year>). <article-title>Acoustic control by wave field synthesis</article-title>. <source>J. Acoust. Soc. Am.</source> <volume>93</volume>, <fpage>2764</fpage>&#x2013;<lpage>2778</lpage>. <pub-id pub-id-type="doi">10.1121/1.405852</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bosi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Servetti</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chafe</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Rottondi</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Experiencing remote classical music performance over long distance: a JackTrip concert between two continents during the pandemic</article-title>. <source>J. Audio Eng. Soc.</source> <volume>69</volume>, <fpage>934</fpage>&#x2013;<lpage>945</lpage>. <pub-id pub-id-type="doi">10.17743/jaes.2021.0056</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>C&#xe1;ceres</surname>
<given-names>J.-P.</given-names>
</name>
<name>
<surname>Chafe</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2010a</year>). <article-title>JackTrip: under the hood of an engine for network audio</article-title>. <source>J. New Music Res.</source> <volume>39</volume>, <fpage>183</fpage>&#x2013;<lpage>187</lpage>. <pub-id pub-id-type="doi">10.1080/09298215.2010.481361</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>C&#xe1;ceres</surname>
<given-names>J.-P.</given-names>
</name>
<name>
<surname>Chafe</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2010b</year>). <article-title>JackTrip/SoundWIRE meets server farm</article-title>. <source>Comput. Music J.</source> <volume>34</volume>, <fpage>29</fpage>&#x2013;<lpage>34</lpage>. <pub-id pub-id-type="doi">10.1162/comj_a_00001</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Car&#xf4;t</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hohn</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Werner</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2009</year>). &#x201c;<article-title>Netjack &#x2013; remote music collaboration with electronic sequencers on the Internet</article-title>,&#x201d; in <source>
<italic>Proceedings of the 7th Linux audio conference</italic> (Parma, Italy)</source>.</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chafe</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>I am streaming in a room</article-title>. <source>Front. Digital Humanit.</source> <volume>5</volume>. <pub-id pub-id-type="doi">10.3389/fdigh.2018.00027</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chafe</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Oshiro</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Jacktrip on Raspberry Pi</article-title>,&#x201d; in <source>Proceedings of the Linux audio conference 2019</source> (<publisher-loc>Stanford, CA, USA: CCRMA</publisher-loc>: <publisher-name>Stanford University</publisher-name>).</citation>
</ref>
<ref id="B18">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chafe</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wilson</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Leistikow</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Chisholm</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Scavone</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2000</year>). &#x201c;<article-title>A simplified approach to high quality music and sound over IP</article-title>,&#x201d; in <source>
<italic>Proceedings of the COST G-6 Conference on digital audio effects (DAFX-00)</italic> (Verona, Italy)</source>.</citation>
</ref>
<ref id="B19">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chafe</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wilson</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Walling</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2002</year>). &#x201c;<article-title>Physical model synthesis with application to Internet acoustics</article-title>,&#x201d; in <source>
<italic>2002 IEEE international Conference on acoustics, speech, and signal processing</italic> (Orlando, FL, USA: IEEE)</source>, <fpage>IV&#x2013;4056&#x2013;IV</fpage>&#x2013;<lpage>4059</lpage>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cohen</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>1977</year>). <article-title>Specifications for the network voice protocol (NVP)</article-title>. <source>Tech. Rep. RFC0741</source>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cohen</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>1981</year>). <article-title>On holy wars and a plea for Peace</article-title>. <source>Computer</source> <volume>14</volume>, <fpage>48</fpage>&#x2013;<lpage>54</lpage>. <pub-id pub-id-type="doi">10.1109/c-m.1981.220208</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Correll</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Barendt</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2005</year>). &#x201c;<article-title>Design considerations for software only implementations of the IEEE 1588 precision time protocol</article-title>,&#x201d; in <source>
<italic>Proceedings of the IEEE 1588 conference</italic> (Winterthur, Switzerland)</source>.</citation>
</ref>
<ref id="B23">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Daniel</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Moreau</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Nicol</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2003</year>). &#x201c;<article-title>Further investigations of high-order ambisonics and wavefield synthesis for holophonic sound imaging</article-title>,&#x201d; in <source>114th convention of the</source>, <volume>114</volume>. <publisher-loc>Amsterdam, Netherlands</publisher-loc>: <publisher-name>Audio Engineering Society</publisher-name>.</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<collab>Dante</collab> (<year>2022</year>). <article-title>What is Dante?</article-title> <source>Audinate &#x7c; Dante Pro Av. Netw.</source> <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://www.audinate.com/meet-dante/what-is-dante">https://www.audinate.com/meet-dante/what-is-dante</ext-link> (Accessed</comment> <comment>November 01, 2024)</comment>.</citation>
</ref>
<ref id="B25">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>de Bruijn</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2004</year>). <source>Application of wave field synthesis in videoconferencing</source>. <publisher-loc>Delft, Netherlands</publisher-loc>: <publisher-name>Technische Universiteit Delft</publisher-name>. <comment>Ph.D. thesis</comment>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>De Poli</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Rocchesso</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>Physically based sound modelling</article-title>. <source>Organised Sound.</source> <volume>3</volume>, <fpage>61</fpage>&#x2013;<lpage>76</lpage>. <pub-id pub-id-type="doi">10.1017/s1355771898009182</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Devonport</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Foss</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>The distribution of ambisonic and point source rendering to ethernet AVB speakers</article-title>,&#x201d; in <source>
<italic>Proceedings of ICSA 2019</italic> (ilmenau, Germany)</source>.</citation>
</ref>
<ref id="B28">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Drioli</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Allocchio</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Buso</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2013</year>). &#x201c;<article-title>Networked performances and natural interaction via LOLA: low latency high quality A/V streaming system</article-title>,&#x201d; in <source>Conference proceedings of the second international conference on information technologies for performing arts, media access and entertainment, ECLAP</source> (<publisher-loc>Porto, Portugal</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>240</fpage>&#x2013;<lpage>250</lpage>.</citation>
</ref>
<ref id="B29">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Edison</surname>
<given-names>J. C.</given-names>
</name>
<name>
<surname>Fischer</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>White</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2002</year>). &#x201c;<article-title>IEEE-1588 standard for a precision clock synchronization protocol for networked measurement and control systems</article-title>,&#x201d; in <source>Proceedings of the 34th annual precise time and time interval systems and applications meeting</source> (<publisher-loc>Reston, VA, USA</publisher-loc>).</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fischer</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Case study: performing band rehearsals on the internet with Jamulus</article-title>
</citation>
</ref>
<ref id="B31">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Frank</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zotter</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Sontacchi</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Producing 3D audio in ambisonics</article-title>,&#x201d; in <source>Audio engineering society 57th international conference</source> (<publisher-loc>Hollywood, CA, USA</publisher-loc>).</citation>
</ref>
<ref id="B32">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Gabrielli</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Squartini</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Principi</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Piazza</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2012</year>). &#x201c;<article-title>Networked Beagleboards for wireless music applications</article-title>,&#x201d; in <source>
<italic>Proceedings of the 5th European DSP Education and research conference</italic> (amsterdam, The Netherlands)</source>, <fpage>291</fpage>&#x2013;<lpage>295</lpage>.</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Geier</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ahrens</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Spors</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Object-based audio reproduction and the audio scene description format</article-title>. <source>Organised Sound.</source> <volume>15</volume>, <fpage>219</fpage>&#x2013;<lpage>227</lpage>. <pub-id pub-id-type="doi">10.1017/s1355771810000324</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Grani</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Di Carlo</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Madrid Portillo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Girardi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Paisa</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Banas</surname>
<given-names>J. S.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). &#x201c;<article-title>Gestural control of wavefield synthesis</article-title>,&#x201d; in <source>
<italic>Sound and music computing conference proceedings</italic> (hamburg, Germany)</source>.</citation>
</ref>
<ref id="B35">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Hardman</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Sasse</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Handley</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Watson</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>1995</year>). &#x201c;<article-title>Reliable audio for use over the internet</article-title>,&#x201d; in <conf-name>Proceedings of INET&#x2019;95 (Honolulu, Hawaii: The Internet Society)</conf-name>, <fpage>171</fpage>&#x2013;<lpage>178</lpage>.<volume>95</volume>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hardman</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Sasse</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Kouvelas</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>Successful multiparty audio communication over the Internet</article-title>. <source>Commun. ACM</source> <volume>41</volume>, <fpage>74</fpage>&#x2013;<lpage>80</lpage>. <pub-id pub-id-type="doi">10.1145/274946.274959</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hildebrand</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>AES67-2013: AES standard for audio applications of networks - high-performance streaming audio-over-IP interoperability</article-title>,&#x201d; in <source>
<italic>Proceedings of the NAB broadcast engineering conference</italic> (Las Vegas, NV, USA)</source>.</citation>
</ref>
<ref id="B38">
<citation citation-type="book">
<collab>IEEE</collab> (<year>2011</year>). &#x201c;<article-title>IEEE Std 802.1BA-2011, IEEE standard for local and metropolitan area networks&#x2014;audio Video bridging (AVB) systems</article-title>,&#x201d; in <source>Tech. rep.</source> <publisher-name>IEEE</publisher-name>.</citation>
</ref>
<ref id="B39">
<citation citation-type="book">
<collab>IEEE</collab> (<year>2018</year>). &#x201c;<article-title>IEEE standard for ethernet (IEEE Std 802.3&#x2122;-2018 revision of IEEE Std 802.3-2015)</article-title>,&#x201d; in <source>Tech. rep.</source> <publisher-name>IEEE</publisher-name>.</citation>
</ref>
<ref id="B40">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kaiser</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2011</year>). &#x201c;<article-title>Transaural Audio - the reproduction of binaural signals over loudspeakers</article-title>,&#x201d; in <source>Universit&#xe4;t f&#xfc;r Musik und darstellende Kunst</source>. <publisher-loc>Graz, Austria</publisher-loc>: <publisher-name>Graz/Institut f&#xfc;r Elekronische Musik und Akustik/IRCAM</publisher-name>. <comment>Ph.D. thesis</comment>.</citation>
</ref>
<ref id="B41">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Kerrisk</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>socket(2) - Linux manual page</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://man7.org/linux/man-pages/man2/socket.2.html">https://man7.org/linux/man-pages/man2/socket.2.html</ext-link> (Accessed November 01, 2024)</comment>.</citation>
</ref>
<ref id="B42">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kshemkalyani</surname>
<given-names>A. D.</given-names>
</name>
<name>
<surname>Singhal</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2011</year>). <source>Distributed computing: principles, algorithms, and systems</source>. <publisher-name>Cambridge University Press</publisher-name>.</citation>
</ref>
<ref id="B43">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lago</surname>
<given-names>N. P.</given-names>
</name>
<name>
<surname>Kon</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2003</year>). &#x201c;<article-title>A middleware system for distributed real-time multimedia processing</article-title>,&#x201d; in <source>Proceedings of the IX Brazilian symposium on multimedia systems and the WEB</source>.</citation>
</ref>
<ref id="B44">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Lopez-Lezcano</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2012</year>). &#x201c;<article-title>From Jack to UDP packets to sound and back</article-title>,&#x201d; in <conf-name>10th International Linux Audio Conference</conf-name> (<publisher-name>Stanford, CA, USA: CCRMA, Stanford University</publisher-name>).</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Marouani</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Dagenais</surname>
<given-names>M. R.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Internal clock drift estimation in computer clusters</article-title>. <source>J. Comput. Netw. Commun.</source> <volume>2008</volume>, <fpage>e583162</fpage>. <pub-id pub-id-type="doi">10.1155/2008/583162</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Meyer</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Cotton</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Vegoda</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2010</year>). <source>
<italic>IANA Guidelines for IPv4 multicast address assignments</italic>. Request for comments RFC 5771</source>. <publisher-name>Internet Engineering Task Force</publisher-name>.</citation>
</ref>
<ref id="B47">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Michon</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Orlarey</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Letz</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Fober</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Real time audio digital signal processing with faust and the teensy</article-title>,&#x201d; in <source>
<italic>Proceedings of the Sound and music computing conference (SMC-19)</italic> (malaga, Spain)</source>.</citation>
</ref>
<ref id="B48">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Michon</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Orlarey</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Letz</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Fober</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Roosenburg</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Embedded real-time audio signal processing with faust</article-title>,&#x201d; in <source>
<italic>Proceedings of the international faust conference (IFC-20)</italic> (paris, France)</source>.</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mitterhuber</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sharafi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Tom&#xe1;s</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Ottosonics</article-title>. <source>Tangible Music Lab.</source> <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://tamlab.kunstuni-linz.at/projects/ottosonics/">https://tamlab.kunstuni-linz.at/projects/ottosonics/</ext-link>(Accessed</comment> <comment>November 01, 2024)</comment>.</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mueller</surname>
<given-names>R. K.</given-names>
</name>
</person-group> (<year>1971</year>). <article-title>Acoustic holography</article-title>. <source>Proc. IEEE</source> <volume>59</volume>, <fpage>1319</fpage>&#x2013;<lpage>1335</lpage>. <pub-id pub-id-type="doi">10.1109/proc.1971.8407</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Nicol</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Sound field</article-title>,&#x201d; in <source>Immersive sound</source> (<publisher-loc>NY, USA</publisher-loc>: <publisher-name>Routledge</publisher-name>), <fpage>276</fpage>&#x2013;<lpage>310</lpage>.</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Orlarey</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fober</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Letz</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>FAUST: an efficient functional approach to DSP programming</article-title>. <source>New Comput. paradigms Comput. music</source>, <fpage>65</fpage>&#x2013;<lpage>96</lpage>.</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pulkki</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Virtual sound source positioning using vector base amplitude panning</article-title>. <source>J. Audio Eng. Soc.</source> <volume>45</volume>, <fpage>456</fpage>&#x2013;<lpage>466</lpage>.</citation>
</ref>
<ref id="B54">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Renaud</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Car&#xf4;t</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rebelo</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2007</year>). &#x201c;<article-title>Networked music performance: state of the art</article-title>,&#x201d; in <source>
<italic>30th AES international Conference on intelligent audio environments</italic> (saariselk&#xe4;, Finland)</source>.</citation>
</ref>
<ref id="B55">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Rushton</surname>
<given-names>T. A.</given-names>
</name>
<name>
<surname>Michon</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Letz</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>A microcontroller-based network client towards distributed spatial audio</article-title>,&#x201d; in <source>
<italic>Proceedings of the Sound and music computing conference (SMC-23)</italic> (stockholm, Sweden)</source>.</citation>
</ref>
<ref id="B56">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sacchetto</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Servetti</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chafe</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>JackTrip-WebRTC: networked music experiments with PCM stereo audio in a Web browser</article-title>,&#x201d; in <source>Proceedings of the International web audio Conference</source> (<publisher-name>Barcelona, Spain: UPF</publisher-name>). <comment>WAC &#x2019;21</comment>.</citation>
</ref>
<ref id="B57">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Schiavoni</surname>
<given-names>F. L.</given-names>
</name>
<name>
<surname>Queiroz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wanderley</surname>
<given-names>M. M.</given-names>
</name>
</person-group> (<year>2013</year>). &#x201c;<article-title>Alternatives in network transport protocols for audio streaming applications</article-title>,&#x201d; in <source>
<italic>Proceedings of the international computer music conference</italic> (perth, Australia)</source>.</citation>
</ref>
<ref id="B58">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Schulzrinne</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>1992</year>). &#x201c;<article-title>Voice communication across the Internet: a network voice terminal</article-title>,&#x201d;. <publisher-loc>Amherst, MA, USA</publisher-loc>: <publisher-name>University of Massachusetts at Amherst, Department of Computer and Information Science</publisher-name>.</citation>
</ref>
<ref id="B59">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tongzhou</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Lunhui</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Research and implementation of high precision clock synchronization of network audio system based on FPGA and 10-gigabit ethernet</article-title>,&#x201d; in <source>Proceedings of the 5th international conference on information systems and computer aided education (ICISCAE)</source> (<publisher-loc>China: IEEE</publisher-loc>: <publisher-name>Dalian</publisher-name>), <fpage>154</fpage>&#x2013;<lpage>161</lpage>.</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Turchet</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Fischione</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Elk audio OS: an open source operating system for the internet of musical things</article-title>. <source>ACM Trans. Internet Things</source> <volume>2</volume> (<issue>12</issue>), <fpage>1</fpage>&#x2013;<lpage>18</lpage>. <pub-id pub-id-type="doi">10.1145/3446393</pub-id>
</citation>
</ref>
<ref id="B61">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Turchet</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Tomasetti</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Immersive networked music performance systems: identifying latency factors</article-title>,&#x201d; in <source>
<italic>2023 Immersive and 3D audio: from Architecture to automotive</italic> (bologna, Italy: IEEE)</source>.</citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Turletti</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>1994</year>). <article-title>The INRIA videoconferencing system (IVS)</article-title>. <source>ConeXions</source> <volume>8</volume>.</citation>
</ref>
<ref id="B63">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Verheijen</surname>
<given-names>E. N. G.</given-names>
</name>
</person-group> (<year>1998</year>). <source>
<italic>Sound Reproduction by wave field synthesis</italic>. Ph.D. Thesis</source>. <publisher-loc>Delft, Netherlands</publisher-loc>: <publisher-name>Technical University Delft</publisher-name>.</citation>
</ref>
<ref id="B64">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Winter</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ahrens</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Spors</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>A geometric model for spatial aliasing in wave field synthesis</article-title>,&#x201d; in <source>Proceedings of the German annual conference on acoustics (DAGA)</source> (<publisher-loc>Munich, Germany</publisher-loc>).</citation>
</ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Woszczyk</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Settel</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Pennycook</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Rowe</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Galanter</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2000</year>). <article-title>Real-time streaming of multichannel audio data over internet</article-title>. <source>J. Audio Eng. Soc.</source> <volume>48</volume>, <fpage>627</fpage>&#x2013;<lpage>641</lpage>.</citation>
</ref>
<ref id="B66">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ziemer</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Wave field synthesis</article-title>,&#x201d; in <source>Psychoacoustic music sound field synthesis</source> <publisher-name>Current Research in Systematic Musicology. Publishing Springer International</publisher-name>, <fpage>203</fpage>&#x2013;<lpage>243</lpage>.</citation>
</ref>
</ref-list>
</back>
</article>