<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="brief-report" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neuroinform.</journal-id>
<journal-title>Frontiers in Neuroinformatics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neuroinform.</abbrev-journal-title>
<issn pub-type="epub">1662-5196</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fninf.2023.1250260</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Brief Research Report</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>PyDapsys: an open-source library for accessing electrophysiology data recorded with DAPSYS</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Konradi</surname>
<given-names>Peter</given-names>
</name>
<xref rid="aff1" ref-type="aff"><sup>1</sup></xref>
<xref rid="c001" ref-type="corresp"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2332392/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Troglio</surname>
<given-names>Alina</given-names>
</name>
<xref rid="aff2" ref-type="aff"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1908472/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>P&#x00E9;rez Garriga</surname>
<given-names>Ariadna</given-names>
</name>
<xref rid="aff1" ref-type="aff"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2415287/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>P&#x00E9;rez Mart&#x00ED;n</surname>
<given-names>Aar&#x00F3;n</given-names>
</name>
<xref rid="aff3" ref-type="aff"><sup>3</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/434278/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>R&#x00F6;hrig</surname>
<given-names>Rainer</given-names>
</name>
<xref rid="aff1" ref-type="aff"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Namer</surname>
<given-names>Barbara</given-names>
</name>
<xref rid="aff2" ref-type="aff"><sup>2</sup></xref>
<xref rid="aff4" ref-type="aff"><sup>4</sup></xref>
<xref rid="aff5" ref-type="aff"><sup>5</sup></xref>
<xref rid="fn0005" ref-type="author-notes"><sup>&#x2020;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/456801/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Kutafina</surname>
<given-names>Ekaterina</given-names>
</name>
<xref rid="aff1" ref-type="aff"><sup>1</sup></xref>
<xref rid="fn0005" ref-type="author-notes"><sup>&#x2020;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1609793/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Institute of Medical Informatics, Medical Faculty, RWTH Aachen University</institution>, <addr-line>Aachen</addr-line>, <country>Germany</country></aff>
<aff id="aff2"><sup>2</sup><institution>Research Group Neuroscience, IZKF, RWTH Aachen</institution>, <addr-line>Aachen</addr-line>, <country>Germany</country></aff>
<aff id="aff3"><sup>3</sup><institution>Simulation and Data Lab Neuroscience, J&#x00FC;lich Supercomputing Centre (JSC), Institute for Advanced Simulation, JARA, Forschungszentrum J&#x00FC;lich GmbH</institution>, <addr-line>J&#x00FC;lich</addr-line>, <country>Germany</country></aff>
<aff id="aff4"><sup>4</sup><institution>Department for Neurophysiology, University Hospital RWTH Aachen</institution>, <addr-line>Aachen</addr-line>, <country>Germany</country></aff>
<aff id="aff5"><sup>5</sup><institution>Institute of Physiology and Pathophysiology, Friedrich-Alexander-Universit&#x00E4;t Erlangen-N&#x00FC;rnberg</institution>, <addr-line>Erlangen</addr-line>, <country>Germany</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0006">
<p>Edited by: Christian Haselgrove, UMass Chan Medical School, United States</p>
</fn>
<fn fn-type="edited-by" id="fn0007">
<p>Reviewed by: Rania Mohamed Hassan Baleela, University of Khartoum, Sudan; Candido Cabo, The City University of New York, United States</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Peter Konradi, <email>peter.konradi@rwth-aachen.de</email></corresp>
<fn fn-type="equal" id="fn0005">
<p><sup>&#x2020;</sup>These authors have contributed equally to this work</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>14</day>
<month>09</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>17</volume>
<elocation-id>1250260</elocation-id>
<history>
<date date-type="received">
<day>29</day>
<month>06</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>28</day>
<month>08</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2023 Konradi, Troglio, P&#x00E9;rez Garriga, P&#x00E9;rez Mart&#x00ED;n, R&#x00F6;hrig, Namer and Kutafina.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Konradi, Troglio, P&#x00E9;rez Garriga, P&#x00E9;rez Mart&#x00ED;n, R&#x00F6;hrig, Namer and Kutafina</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>In the field of neuroscience, a considerable number of commercial data acquisition and processing solutions rely on proprietary formats for data storage. This often leads to data being locked up in formats that are only accessible by using the original software, which may lead to interoperability problems. In fact, even the loss of data access is possible if the software becomes unsupported, changed, or otherwise unavailable. To ensure FAIR data management, strategies should be established to enable long-term, independent, and unified access to data in proprietary formats. In this work, we demonstrate PyDapsys, a solution to gain open access to data that was acquired using the proprietary recording system DAPSYS. PyDapsys enables us to open the recorded files directly in Python and saves them as NIX files, commonly used for open research in the electrophysiology domain. Thus, PyDapsys secures efficient and open access to existing and prospective data. The manuscript demonstrates the complete process of reverse engineering a proprietary electrophysiological format on the example of microneurography data collected for studies on pain and itch signaling in peripheral neural fibers.</p>
</abstract>
<kwd-group>
<kwd>interoperability</kwd>
<kwd>open data</kwd>
<kwd>FAIR</kwd>
<kwd>data management tools</kwd>
<kwd>reverse-engineered</kwd>
<kwd>microneurography</kwd>
<kwd>electrophysiology</kwd>
<kwd>pain</kwd>
</kwd-group>
<counts>
<fig-count count="2"/>
<table-count count="2"/>
<equation-count count="0"/>
<ref-count count="16"/>
<page-count count="6"/>
<word-count count="4325"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1.</label>
<title>Introduction</title>
<p>Many commercial software solutions use custom proprietary formats to store their data. Reasons vary from dealing with special use cases to trying to lock users into a vendor-specific ecosystem. While there is a trend in the general IT space to open-source custom solutions and establish cross-vendor standards (<xref ref-type="bibr" rid="ref8">Kilamo et al., 2012</xref>), in the scientific world, the focus is put on FAIR data principles (<xref ref-type="bibr" rid="ref17">Wilkinson et al., 2016</xref>). FAIR principles consist of a number of requirements for data to be findable, accessible, interoperable, and reusable. Proprietary formats naturally obstruct the adoption of these principles. In some research domains, large efforts are put into building solutions to convert proprietary formats into open standards, while simultaneously lobbying companies to use open formats. Examples of such formats are DICOM<xref rid="fn0001" ref-type="fn"><sup>1</sup></xref> for storing, managing, and exchanging medical images and EDF (<xref ref-type="bibr" rid="ref7">Kemp et al., 1992</xref>) for biosignals, including EEG systems.</p>
<p>Progress has also been made in the field of neuroscience, where the Neuroscience Information Exchange format (NIX) (<xref ref-type="bibr" rid="ref13">Stoewer et al., 2014</xref>) and the Neurodata Without Borders (NWB) (<xref ref-type="bibr" rid="ref11">R&#x00FC;bel et al., 2022</xref>) projects are aiming to establish community standards for sharing neuroscientific data. Both projects specify a storage layout, which is implemented on top of the Hierarchical Data Format (HDF5) but use different approaches to model data. NWB uses a stricter and more standardized data model, whereas NIX allows for a comparatively flexible structure and can describe the file contents using the open metadata Markup Language (odML) (<xref ref-type="bibr" rid="ref5">Grewe et al., 2011</xref>).</p>
<p>However, smaller fields of neuroscience are facing challenges to fully adopt FAIR data principles, as vendors may not have the resources to address the specific wishes of such small user-bases. The &#x201C;Data Acquisition Processor System&#x201D; (DAPSYS)<xref rid="fn0002" ref-type="fn"><sup>2</sup></xref> is a general-purpose neurophysiological data acquisition system (DAS) for recording and processing neural signals, which is, among other places, used in the microneurography (MNG) lab of the University Hospital RWTH Aachen. MNG is an electrophysiological technique to record activity from single nerve fibers of the peripheral nervous system using a single microelectrode (<xref ref-type="bibr" rid="ref16">Vallbo and Hagbarth, 1968</xref>; <xref ref-type="bibr" rid="ref001">Torebjork and Hallin, 1974</xref>; <xref ref-type="bibr" rid="ref1">Ackerley and Watkins, 2018</xref>). Due to the small size of the electrode, the method causes only minimal discomfort and does not require anesthetics. This means that the volunteer stays awake and cooperates during the recording, making it possible to correlate nerve fiber signals with individual sensations. Thus, MNG is a unique translational method in sensory research in humans, especially in chronic pain and itch.</p>
<p>DAPSYS uses a proprietary format to store data and only offers manual (file-by-file) export of the recordings to CSV files. However, the CSV exports produce comparatively large files (see <xref rid="tab1" ref-type="table">Table 1</xref>) and take a long time (see <xref rid="tab2" ref-type="table">Table 2</xref>). In addition, some minor precision loss due to the fixed number of decimals in the exported CSV is observed. Our recent works on establishing data-sharing standards in the MNG community and developing a computational pipeline for spike analysis in MNG data (<xref ref-type="bibr" rid="ref12">Schlebusch et al., 2021</xref>; <xref ref-type="bibr" rid="ref10">Kutafina et al., 2022</xref>; <xref ref-type="bibr" rid="ref14">Troglio et al., 2023</xref>) has raised the urgency for an efficient way to read DAPSYS recordings and store them in more suitable data formats, such as HDF5.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>The size difference between CSV files created by the DAPSYS export and the files created by using PyDapsys with the NIX-exporter of the Neo library.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="center" valign="top">Original file size [MiB]</th>
<th align="center" valign="top">CSV file size [MiB]</th>
<th align="center" valign="top">NIX/H5 file size [MiB]</th>
<th align="center" valign="top">Size increase CSV [%]</th>
<th align="center" valign="top">Size increase NIX/H5 [%]</th>
</tr>
</thead>
<tbody>
<tr>
<td align="center" valign="top">44.5</td>
<td align="center" valign="top">229.5</td>
<td align="center" valign="top">45.1</td>
<td align="center" valign="top">415.73</td>
<td align="center" valign="top">1.35</td>
</tr>
<tr>
<td align="center" valign="top">109.0</td>
<td align="center" valign="top">576.2</td>
<td align="center" valign="top">109.7</td>
<td align="center" valign="top">428.62</td>
<td align="center" valign="top">0.64</td>
</tr>
<tr>
<td align="center" valign="top">124.9</td>
<td align="center" valign="top">664.5</td>
<td align="center" valign="top">126.1</td>
<td align="center" valign="top">432.83</td>
<td align="center" valign="top">0.96</td>
</tr>
<tr>
<td align="center" valign="top">165.8</td>
<td align="center" valign="top">889.2</td>
<td align="center" valign="top">167.4</td>
<td align="center" valign="top">436.31</td>
<td align="center" valign="top">0.97</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>The time comparison for exporting the continuous recording.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="center" valign="top">Original file size [MiB]</th>
<th align="center" valign="top">CSV export time&#x002A; [s]</th>
<th align="center" valign="top">PyDapsys export to NIX/H5 time&#x002A; [s]</th>
<th align="center" valign="top">Speedup PyDapsys vs. CSV export&#x002A;</th>
<th align="center" valign="top">PyDapsys total time [s]</th>
</tr>
</thead>
<tbody>
<tr>
<td align="center" valign="top">44.5</td>
<td align="center" valign="top">35</td>
<td align="center" valign="top">0.36</td>
<td align="center" valign="top">97.2</td>
<td align="center" valign="top">0.77</td>
</tr>
<tr>
<td align="center" valign="top">109.0</td>
<td align="center" valign="top">91</td>
<td align="center" valign="top">0.46</td>
<td align="center" valign="top">197.8</td>
<td align="center" valign="top">0.84</td>
</tr>
<tr>
<td align="center" valign="top">124.9</td>
<td align="center" valign="top">102</td>
<td align="center" valign="top">0.68</td>
<td align="center" valign="top">150.0</td>
<td align="center" valign="top">1.03</td>
</tr>
<tr>
<td align="center" valign="top">165.8</td>
<td align="center" valign="top">133</td>
<td align="center" valign="top">0.98</td>
<td align="center" valign="top">135.7</td>
<td align="center" valign="top">1.49</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>&#x002A;Time required for processing and writing, excluding user interaction.</p>
</table-wrap-foot>
</table-wrap>
<p>While there are many commercial applications for reverse engineering, most of them target computer science professionals and the primary use-case of reverse engineering software, not file formats. The MARBLE project<xref rid="fn0003" ref-type="fn"><sup>3</sup></xref> is to our best knowledge the first research-oriented solution to reverse engineer file formats with the aim of making the process as accessible as possible. However, at the time of the reported work, MARBLE was still in development and the usage required problem-specific adjustments.</p>
<p>Therefore, in this paper, we show our approach to reverse engineering the DAPSYS file format and implement a Python library to gain open access to our own data recorded in the microneurography lab. By providing functionality to load data into the structure defined by the Neo library (<xref ref-type="bibr" rid="ref4">Garcia et al., 2014</xref>), it can be simply exported to multiple data formats used in electrophysiology, including NIX. This ensures full access to the data even if DAPSYS is unavailable.</p>
<p>The primary aim of our work is to ensure the accessibility and interoperability of DAPSYS-recorded data sets. The secondary aim is to share the steps of our reverse engineering solution with the neuroscience community to support building FAIR access to rare data formats.</p>
</sec>
<sec sec-type="method" id="sec2">
<label>2.</label>
<title>Method</title>
<sec id="sec3">
<label>2.1.</label>
<title>Data</title>
<p>We used four DAPSYS files, recorded at the microneurography labs of the University Hospital RWTH Aachen and Friedrich-Alexander-University of Erlangen-N&#x00FC;rnberg. The studies involving human participants were reviewed and approved by the Ethics Boards of those two institutions with the corresponding numbers EK141-19 and 4361. The participants provided their written informed consent, and the studies were conducted according to the Declaration of Helsinki.</p>
</sec>
<sec id="sec4">
<label>2.2.</label>
<title>Reverse engineering method</title>
<p>For the reverse engineering process, we used the hex editor &#x201C;ImHex&#x201D;<xref rid="fn0004" ref-type="fn"><sup>4</sup></xref> to open and analyze the DAPSYS files. A hex editor shows the binary contents of a file in hexadecimal representation. A value of a single byte can be represented by only two characters, making it easier to recognize patterns (see <xref rid="fig1" ref-type="fig">Figure 1A</xref> for an example). Since we knew what values the file should contain, we were able to search for them and identify related fields. From there on, we identified structures based on repeating patterns.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p><bold>(A)</bold> &#x201C;ImHex&#x201D; showing the start of a DAPSYS files&#x2019; table of contents section. Shown are the hexadecimal byte-values of the respective address and the interpretation of that byte as characters. Data fields of structures are shown in different colors. The address shown is in relation to the start of the table of contents. <bold>(B)</bold> Structure of the same file from panel <bold>(A)</bold> shown by the GUI of DAPSYS.</p>
</caption>
<graphic xlink:href="fninf-17-1250260-g001.tif"/>
</fig>
<p>The functions of data fields in the structures were then identified by using the following workflow:</p>
<list list-type="order">
<list-item><p>Make changes to the file using the DAPSYS GUI (for example: changing the plot configuration, removing data points, etc.).</p></list-item>
<list-item><p>Track these changes DAPSYS made to the binary file and identify changed fields in the hex editor.</p></list-item>
<list-item><p>Open a different recording in the hex editor, identify the known fields, and change their values using the hex editor.</p></list-item>
<list-item><p>Open the changed file from step 3 in DAPSYS and verify that the changes made to the recording fit with the assumed function of the field.</p></list-item>
</list>
<p>This process was substantially supported by built-in &#x201C;ImHex&#x201D; functions like the pattern language that can be used to specify the layout of structures in the binary file. These structures can be utilized to highlight and verify known structures and fields in the file.</p>
</sec>
<sec id="sec5">
<label>2.3.</label>
<title>Concept of the library implementation</title>
<p>Based on the results from the reverse-engineering process, we implemented a Python library capable of opening and processing recordings. The library also offers a method to export data from DAPSYS recordings into HDF5 files using the NIX structure (abbreviated as NIX/H5) for easier data exchange between labs and software.</p>
<sec id="sec6">
<label>2.3.1.</label>
<title>Verification</title>
<p>To verify the implementation of the file format in PyDapsys, we read each of the four DAPSYS files (see 2.1) with PyDapsys. The read values were then compared to the CSV files. As the values in the exported CSV files only have limited precision (6 or 4 decimal places, depending on the type of data exported), we first rounded the values read from the file to the same precision before comparing them. Comparison of floating-point values was done by comparing the absolute difference of two values to the system epsilon for 64-bit floating point (f64) values. Numeric values from the CSV were converted to f64 values using built-in Python functions. When comparing f64 with 32-bit floating point (f32) values, the f32 values were first converted to f64. Texts were compared with built-in Python functions.</p>
</sec>
<sec id="sec7">
<label>2.3.2.</label>
<title>Performance testing</title>
<p>We also compared the performance (duration and file sizes) of the CSV export of DAPSYS and the export to NIX/H5 using PyDapsys. To achieve comparable measurements, we only looked at the time each system required to write the continuous recording to their respective target format, without the time required for user interactions or loading the data. We had to focus on a single data stream, as DAPSYS would require user interactions in between exporting multiple streams. We chose to focus on the continuous recording, as it makes up the largest part of a file&#x2019;s size. We also excluded loading times, as there was no reliable way to measure them for DAPSYS. Times for PyDapsys were measured using the wall-clock time directly in the Python program, whereas DAPSYS times were taken by a stopwatch. All measurements were performed on the same system.</p>
</sec>
</sec>
</sec>
<sec sec-type="results" id="sec8">
<label>3.</label>
<title>Results</title>
<sec id="sec9">
<label>3.1.</label>
<title>Analysis of the DAPSYS file structure</title>
<p>The DAPSYS user interface displays the contents of a file in a hierarchical structure, composed of folders, text streams, and data streams (see <xref rid="fig1" ref-type="fig">Figure 1B</xref>). DAPSYS binary files store data in a flat structure that can be split into 4 parts:</p>
<list list-type="order">
<list-item><p>Header. Files begin with a header with a fixed length. Information in the header is not required to read the file contents.</p></list-item>
<list-item><p>Data Pages. DAPSYS stores data in discontinuous chunks, which we call &#x201C;pages.&#x201D; All pages have a unique ID in the context of the file and can hold either data of a waveform or textual data.</p></list-item>
<list-item><p>Table of Contents (ToC). After the last data page, the ToC begins. It defines the hierarchical structure shown in the GUI and comprises of folders, which can have additional child elements and streams. Streams contain an array of data page IDs.</p></list-item>
<list-item><p>Footer. After the ToC, there comes a small footer consisting of a string holding the version and the serial number of the DAPSYS program used to create the file.</p></list-item>
</list>
<sec id="sec10">
<label>3.1.1.</label>
<title>Data pages</title>
<p>As seen in <xref rid="fig2" ref-type="fig">Figure 2</xref>, DAPSYS uses two types of pages: one for waveform data and one for textual data. Both types start with the same fields that store metadata, such as their ID, which is unique among all pages in a file, an identifier for their type (text or waveform), and an optional reference to another page. Waveform pages store the amplitude of the waveform as an array of 32-bit floating point (f32) values, and corresponding timestamps as an array of 64-bit floating point (f64) values. For regularly sampled waveforms, only the first timestamp is saved in the array, while an additional f64 value is used for the regular sampling interval. Text pages are used to store comments as well as sorted spikes. They consist of a string containing the text, and two f64 values. The first f64 value is used to store the timestamp. The second one is used for sorted spikes to indicate the timestamp of the automatically recognized spike. For normal comments, it is set to the same value as the first timestamp. From our observations, DAPSYS writes pages in the order they occur during the recording. If, for example, a comment is entered during a recording, DAPSYS will save the recorded data up to that point in a waveform page, append it to the list of pages followed by the text page containing the comment, and then begin a new waveform page with the new data.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>Simplified logical model of a DAPSYS file. Black arrows show to which other fields the value refers. Large unfilled arrows indicate that an entry extends another one. Gray types define shared header fields for a group of types, with extending arrows indicating the field and the value to identify the group of fields that follow the header (e.g., a &#x201C;Folder&#x201D; and a &#x201C;Stream&#x201D; both start with the fields of an generic &#x201C;Entry.&#x201D; Depending on the value of the field &#x201C;entry_type,&#x201D; the fields following the &#x201C;Entry&#x201D; will be either those of a &#x201C;Stream&#x201D; or a &#x201C;Folder&#x201D;).</p>
</caption>
<graphic xlink:href="fninf-17-1250260-g002.tif"/>
</fig>
</sec>
<sec id="sec11">
<label>3.1.2.</label>
<title>Table of contents</title>
<p>The ToC defines the logical structure of a DAPSYS file. As seen in <xref rid="fig2" ref-type="fig">Figure 2</xref>, its elements consist of folders and streams, all of which have an ID unrelated to the IDs used for pages and a string containing their display name. Folders can have several other elements as children. Streams contain multiple fields for storing the configuration of the plot used to visualize their data and most importantly, contain an array of the page IDs belonging to that stream. A stream may either reference text pages or waveform pages, making it a text or data stream, respectively.</p>
</sec>
</sec>
<sec id="sec12">
<label>3.2.</label>
<title>Development of the Python library &#x201C;PyDapsys&#x201D;</title>
<p>The functionality of PyDapsys [see (<xref ref-type="bibr" rid="ref9">Konradi et al., 2023</xref>) for the repository containing the source code. The package is also available on PyPI as &#x201C;pydapsys&#x201D;] focuses on accessing data stored in a DAPSYS file. Pages are read into a dictionary that maps the page IDs to an object storing the metadata (type of the page, ID, optional ID of the referenced page) and data, i.e., text and timestamps for text pages of the corresponding page. The ToC is represented by folder and stream objects. The folder objects offer dictionary-like access to their children, while stream objects store the IDs of the pages belonging to them. The library uses NumPy (<xref ref-type="bibr" rid="ref6">Harris et al., 2020</xref>) to improve the reading speed and memory efficiency of the arrays storing page IDs, amplitudes, and timestamps. To keep the library portable, NumPy is the only required dependency. The functionality to convert a recording to the Neo structure is implemented as an optional dependency. As different experiment set-ups may produce different structures in the DAPSYS file, there is no &#x201C;universal&#x201D; converter. Instead, the library provides an abstract base class for Neo converters, which offers functions for common conversions (i.e., text stream to event). Based on this class, additional converters may be implemented for different ToC structures.</p>
<sec id="sec13">
<label>3.2.1.</label>
<title>Verification</title>
<p>As described in section 2.3.1, we compared the CSV data exported by DAPSYS with the data read by PyDapsys. Depending on the type of stream being exported to CSV, the resulting file contains different values:</p>
<list list-type="bullet">
<list-item><p>Waveform streams: Contain both the timestamps for each data point and its signal value. Both timestamp and signal values have a precision of 6 decimals.</p></list-item>
<list-item><p>Text streams: Contain the timestamps for each text with a precision of 4 decimals and the text itself.</p></list-item>
</list>
<p>Across all files used for testing, 284,453,786 individual floating-point values were compared, of which 3,009,074 values differed. The maximum difference was 0.00001. As this is exactly the precision offered by waveform CSV-exports, it is most likely a result from rounding errors and not a systemic error in the PyDapsys implementation. There were no differences in the text data.</p>
</sec>
<sec id="sec14">
<label>3.2.2.</label>
<title>Performance testing</title>
<p>As seen in <xref rid="tab1" ref-type="table">Table 1</xref>, storing data in NIX/H5 with Neo had no significant impact on file sizes compared to the original file, whereas the CSV increased the file size by factor 4. PyDapsys reliably outperformed DAPSYS in the time required for exporting a file by more than factor 97 (see <xref rid="tab2" ref-type="table">Table 2</xref>).</p>
</sec>
</sec>
</sec>
<sec sec-type="discussions" id="sec15">
<label>4.</label>
<title>Discussion</title>
<p>In order to make electrophysiological recordings obtained with the DAPSYS DAS available to other systems in our lab, we implemented the open-source Python library &#x201C;PyDapsys.&#x201D; The library has functionality for reading data from DAPSYS files and offers built-in functions to automatically load read data into the structure defined by the Neo library, from where it can be exported to NIX and other data formats, which are used by the neuroscience community and can be read by various other software solutions. By offering direct access to the data stored in DAPSYS files, rounding errors that may occur when exporting the data to CSV are avoided, thus improving the accuracy and quality of subsequent analyses. The library outperforms the DAPSYS CSV export, both in export duration and size of the exported files, while additionally not being dependent on DAPSYS itself. Currently, the usage of the PyDapsys library requires a certain level of programming experience. To make the library available for a more general audience, we are working on implementing a GUI (graphical user interface).</p>
<p>While DAPSYS is not used very commonly, it should be seen as a representative of many domain-specific proprietary formats, which are used in neuroscientific research. FAIR data handling principles require the accessibility and interoperability of data, and the opening of proprietary formats is a necessary step to ensure those qualities (<xref ref-type="bibr" rid="ref2">Berens and Ayhan, 2019</xref>). We expect the presented process of analyzing the files with the &#x201C;ImHex&#x201D; software and modifying the parameters to understand their internal structure to be useful for other research groups, who are facing similar challenges. It is important to note that the DAPSYS file format does not utilize any compression or encryption. Reverse engineering compressed or encrypted data would have made the process significantly more difficult.</p>
<p>In general, our case highlights the importance of proper procedures to ensure long-term access to experimental data. In the microneurography community, the experiments are complex, and many data sets are unique due to rare genetic mutations of the patients. Moreover, guaranteeing reliable access and unification of data also simplifies collaboration between research groups. Therefore, ensuring FAIR principles allows us to optimize the research benefit derived from the data.</p>
<p>The appropriate processes should ideally be put in place early on to ensure that data is available in open formats. For example, if the formats cannot be read using open software, this could include manual exporting new data to open formats once a week to avoid forming a backlog and potentially losing access to large quantities of non-exported data if the original software is not available anymore.</p>
<p>Open science and FAIR principles are becoming more and more widely accepted in academia and in neuroscience in particular. However, at the current stage of ongoing works, it is important to include smaller communities in the discussion, as the popularity of the specific software and hardware solution influences the motivation of the vendors to provide open off-the-shelf solutions. PyDapsys alongside more general emerging approaches, such as MARBLE, serves as an example of a possible solution for these research communities.</p>
</sec>
<sec sec-type="data-availability" id="sec16">
<title>Data availability statement</title>
<p>The data analyzed in this study is subject to the following licenses/restrictions: the hospital regulations limit open data sharing. Data is available upon reasonable request. Requests to access these datasets should be directed to BN, <email>bnamer@ukaachen.de</email>.</p>
</sec>
<sec id="sec17">
<title>Ethics statement</title>
<p>The studies involving humans were approved by the Ethics Boards of the University Hospital RWTH Aachen and Friedrich-Alexander-University of Erlangen-N&#x00FC;rnberg. The studies were conducted in accordance with the local legislation and institutional requirements. The participants provided their written informed consent to participate in this study.</p>
</sec>
<sec id="sec18">
<title>Author contributions</title>
<p>PK developed the software and drafted the manuscript. AT supervised the work on microneurography data. APG and APM supervised the software-development work. RR, BN, and EK supervised the project. All authors substantially revised the manuscript.</p>
</sec>
<sec sec-type="funding-information" id="sec19">
<title>Funding</title>
<p>This work was partially funded by the Excellence Initiative of the German Federal and State Governments G:(DE-82) EXS-SF-SFDdM013 and also supported by the IZKF TN1-6/IA 532006. BN was supported by a grant from the Interdisciplinary Center for Clinical Research within the Faculty of Medicine at the RWTH Aachen University and the German Research Council DFG NA 970 3-1, DFG FOR 2690 project 6.</p>
</sec>
<sec sec-type="COI-statement" id="sec20">
<title>Conflict of interest</title>
<p>APM was employed by Forschungszentrum J&#x00FC;lich GmbH.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="sec100" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<ack>
<p>The authors would like to thank Abigail Morrison for her support and many insightful discussions. Also, they would like to thank Dagmar Krefting for the discussion on interoperability in biosignals.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="ref1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ackerley</surname> <given-names>R.</given-names></name> <name><surname>Watkins</surname> <given-names>R. H.</given-names></name></person-group> (<year>2018</year>). <article-title>Microneurography as a tool to study the function of individual C-Fiber afferents in humans: responses from nociceptors, thermoreceptors, and mechanoreceptors</article-title>. <source>J. Neurophysiol.</source> <volume>120</volume>, <fpage>2834</fpage>&#x2013;<lpage>2846</lpage>. doi: <pub-id pub-id-type="doi">10.1152/jn.00109.2018</pub-id>, PMID: <pub-id pub-id-type="pmid">30256737</pub-id></citation>
</ref>
<ref id="ref2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Berens</surname> <given-names>P.</given-names></name> <name><surname>Ayhan</surname> <given-names>M. S.</given-names></name></person-group> (<year>2019</year>). <article-title>Proprietary data formats block Health Research</article-title>. <source>Nature</source> <volume>565</volume>:<fpage>429</fpage>. doi: <pub-id pub-id-type="doi">10.1038/d41586-019-00231-9</pub-id></citation>
</ref>
<ref id="ref4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Garcia</surname> <given-names>S.</given-names></name> <name><surname>Guarino</surname> <given-names>D.</given-names></name> <name><surname>Jaillet</surname> <given-names>F.</given-names></name> <name><surname>Jennings</surname> <given-names>T.</given-names></name> <name><surname>Pr&#x00F6;pper</surname> <given-names>R.</given-names></name> <name><surname>Rautenberg</surname> <given-names>P. L.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>Neo: an object model for handling electrophysiology data in multiple formats</article-title>. <source>Front. Neuroinform.</source> <volume>8</volume>:<fpage>10</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fninf.2014.00010</pub-id>, PMID: <pub-id pub-id-type="pmid">24600386</pub-id></citation>
</ref>
<ref id="ref5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Grewe</surname> <given-names>J.</given-names></name> <name><surname>Wachtler</surname> <given-names>T.</given-names></name> <name><surname>Benda</surname> <given-names>J.</given-names></name></person-group> (<year>2011</year>). <article-title>A bottom-up approach to data annotation in neurophysiology</article-title>. <source>Front. Neuroinform.</source> <volume>5</volume>:<fpage>16</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fninf.2011.00016</pub-id>, PMID: <pub-id pub-id-type="pmid">21941477</pub-id></citation>
</ref>
<ref id="ref6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Harris</surname> <given-names>C. R.</given-names></name> <name><surname>Jarrod Millman</surname> <given-names>K.</given-names></name> <name><surname>Van Der Walt</surname> <given-names>S. J.</given-names></name> <name><surname>Gommers</surname> <given-names>R.</given-names></name> <name><surname>Virtanen</surname> <given-names>P.</given-names></name> <name><surname>Cournapeau</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Array programming with NumPy</article-title>. <source>Nature</source> <volume>585</volume>, <fpage>357</fpage>&#x2013;<lpage>362</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41586-020-2649-2</pub-id>, PMID: <pub-id pub-id-type="pmid">32939066</pub-id></citation>
</ref>
<ref id="ref7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kemp</surname> <given-names>B.</given-names></name> <name><surname>V&#x00E4;rri</surname> <given-names>A.</given-names></name> <name><surname>Rosa</surname> <given-names>A. C.</given-names></name> <name><surname>Nielsen</surname> <given-names>K. D.</given-names></name> <name><surname>Gade</surname> <given-names>J.</given-names></name></person-group> (<year>1992</year>). <article-title>A simple format for exchange of digitized Polygraphic recordings</article-title>. <source>Electroencephalogr. Clin. Neurophysiol.</source> <volume>82</volume>, <fpage>391</fpage>&#x2013;<lpage>393</lpage>. doi: <pub-id pub-id-type="doi">10.1016/0013-4694(92)90009-7</pub-id>, PMID: <pub-id pub-id-type="pmid">1374708</pub-id></citation>
</ref>
<ref id="ref8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kilamo</surname> <given-names>T.</given-names></name> <name><surname>Hammouda</surname> <given-names>I.</given-names></name> <name><surname>Mikkonen</surname> <given-names>T.</given-names></name> <name><surname>Aaltonen</surname> <given-names>T.</given-names></name></person-group> (<year>2012</year>). <article-title>From proprietary to open source&#x2014;growing an open source ecosystem</article-title>. <source>J. Syst. Softw.</source> <volume>85</volume>, <fpage>1467</fpage>&#x2013;<lpage>1478</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jss.2011.06.071</pub-id></citation>
</ref>
<ref id="ref9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Konradi</surname> <given-names>P.</given-names></name> <name><surname>Troglio</surname> <given-names>A.</given-names></name> <name><surname>Namer</surname> <given-names>B.</given-names></name> <name><surname>Kutafina</surname> <given-names>E.</given-names></name></person-group> (<year>2023</year>). <article-title>Digital-C-Fiber/PyDapsys</article-title>. <source>Zenodo</source>. doi: <pub-id pub-id-type="doi">10.5281/ZENODO.7970520</pub-id></citation>
</ref>
<ref id="ref10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kutafina</surname> <given-names>E.</given-names></name> <name><surname>Troglio</surname> <given-names>A.</given-names></name> <name><surname>De Col</surname> <given-names>R.</given-names></name> <name><surname>R&#x00F6;hrig</surname> <given-names>R.</given-names></name> <name><surname>Rossmanith</surname> <given-names>P.</given-names></name> <name><surname>Namer</surname> <given-names>B.</given-names></name></person-group> (<year>2022</year>). <article-title>Decoding neuropathic pain: can we predict fluctuations of propagation speed in stimulated peripheral nerve?</article-title> <source>Front. Comput. Neurosci.</source> <volume>16</volume>:<fpage>899584</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fncom.2022.899584</pub-id>, PMID: <pub-id pub-id-type="pmid">35966281</pub-id></citation>
</ref>
<ref id="ref11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>R&#x00FC;bel</surname> <given-names>O.</given-names></name> <name><surname>Tritt</surname> <given-names>A.</given-names></name> <name><surname>Ly</surname> <given-names>R.</given-names></name> <name><surname>Dichter</surname> <given-names>B. K.</given-names></name> <name><surname>Ghosh</surname> <given-names>S.</given-names></name> <name><surname>Niu</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>The Neurodata without Borders ecosystem for neurophysiological data science</article-title>. <source>eLife</source> <volume>11</volume>:<fpage>e78362</fpage>. doi: <pub-id pub-id-type="doi">10.7554/eLife.78362</pub-id>, PMID: <pub-id pub-id-type="pmid">36193886</pub-id></citation>
</ref>
<ref id="ref12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schlebusch</surname> <given-names>F.</given-names></name> <name><surname>Kehrein</surname> <given-names>F.</given-names></name> <name><surname>R&#x00F6;hrig</surname> <given-names>R.</given-names></name> <name><surname>Namer</surname> <given-names>B.</given-names></name> <name><surname>Kutafina</surname> <given-names>E.</given-names></name></person-group> (<year>2021</year>). <article-title>openMNGlab: data analysis framework for microneurography &#x2013; a technical report</article-title>. <source>Stud. Health Technol. Inform.</source> <volume>283</volume>, <fpage>165</fpage>&#x2013;<lpage>171</lpage>. doi: <pub-id pub-id-type="doi">10.3233/SHTI210556</pub-id></citation>
</ref>
<ref id="ref13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stoewer</surname> <given-names>A.</given-names></name> <name><surname>Kellner</surname> <given-names>C.</given-names></name> <name><surname>Benda</surname> <given-names>J.</given-names></name> <name><surname>Wachtler</surname> <given-names>T.</given-names></name> <name><surname>Grewe</surname> <given-names>J.</given-names></name></person-group> (<year>2014</year>). <article-title>File format and library for neuroscience data and metadata</article-title>. <source>Front. Neuroinform.</source> <volume>8</volume>:<fpage>15</fpage>. doi: <pub-id pub-id-type="doi">10.3389/conf.fninf.2014.18.00027</pub-id></citation>
</ref>
<ref id="ref14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Troglio</surname> <given-names>A.</given-names></name> <name><surname>Schlebusch</surname> <given-names>F.</given-names></name> <name><surname>R&#x00F6;hrig</surname> <given-names>R.</given-names></name> <name><surname>Dunham</surname> <given-names>J.</given-names></name> <name><surname>Namer</surname> <given-names>B.</given-names></name> <name><surname>Kutafina</surname> <given-names>E.</given-names></name></person-group> (<year>2023</year>). <article-title>odML-tables as a metadata standard in microneurography</article-title>. <source>Stud. Health Technol. Inform.</source> <volume>302</volume>, <fpage>368</fpage>&#x2013;<lpage>369</lpage>. doi: <pub-id pub-id-type="doi">10.3233/SHTI230144</pub-id></citation>
</ref>
<ref id="ref001">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Torebjork</surname> <given-names>H. E.</given-names></name> <name><surname>Hallin</surname> <given-names>R. G.</given-names></name></person-group> (<year>1974</year>). <article-title>Responses in Human A and C Fibres to Repeated Electrical Intradermal Stimulation</article-title>. <source>J. Neur. Neurosurgery Psych.</source> <volume>37</volume>, <fpage>653</fpage>&#x2013;<lpage>64</lpage>. doi: <pub-id pub-id-type="doi">10.1136/jnnp.37.6.653</pub-id>, PMID: <pub-id pub-id-type="pmid">5673644</pub-id></citation>
</ref>
<ref id="ref16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vallbo</surname> <given-names>&#x00C5;. B.</given-names></name> <name><surname>Hagbarth</surname> <given-names>K.-E.</given-names></name></person-group> (<year>1968</year>). <article-title>Activity from skin mechanoreceptors recorded percutaneously in awake human subjects</article-title>. <source>Exp. Neurol.</source> <volume>21</volume>, <fpage>270</fpage>&#x2013;<lpage>289</lpage>. doi: <pub-id pub-id-type="doi">10.1016/0014-4886(68)90041-1</pub-id>, PMID: <pub-id pub-id-type="pmid">5673644</pub-id></citation>
</ref>
<ref id="ref17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wilkinson</surname> <given-names>M. D.</given-names></name> <name><surname>Dumontier</surname> <given-names>M.</given-names></name> <name><surname>Aalbersberg</surname> <given-names>I. J.</given-names></name> <name><surname>Appleton</surname> <given-names>G.</given-names></name> <name><surname>Axton</surname> <given-names>M.</given-names></name> <name><surname>Baak</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>The FAIR guiding principles for scientific data management and stewardship</article-title>. <source>Sci. Data</source> <volume>3</volume>:<fpage>160018</fpage>. doi: <pub-id pub-id-type="doi">10.1038/sdata.2016.18</pub-id>, PMID: <pub-id pub-id-type="pmid">26978244</pub-id></citation>
</ref>
</ref-list>
<fn-group>
<fn id="fn0001">
<p><sup>1</sup>DICOM: Digital Imaging and Communications in Medicine, Medical Imaging Technology Association (MITA), <ext-link xlink:href="https://www.dicomstandard.org" ext-link-type="uri">https://www.dicomstandard.org</ext-link>.</p>
</fn>
<fn id="fn0002">
<p><sup>2</sup>Data Acquisition Processor System (DAPSYS), Brian Turnquist, <ext-link xlink:href="http://dapsys.net" ext-link-type="uri">http://dapsys.net</ext-link>.</p>
</fn>
<fn id="fn0003">
<p><sup>3</sup>MARBLE software project, Steffen Brinckmann et al., <ext-link xlink:href="https://gitlab-public.fz-juelich.de/marble" ext-link-type="uri">https://gitlab-public.fz-juelich.de/marble</ext-link>.</p>
</fn>
<fn id="fn0004">
<p><sup>4</sup>ImHex, Nikolaij &#x201C;WerWolv&#x201D; S&#x00E4;gesser, <ext-link xlink:href="https://github.com/WerWolv/ImHex" ext-link-type="uri">https://github.com/WerWolv/ImHex</ext-link>.</p>
</fn>
</fn-group>
</back>
</article>