<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Public Health</journal-id>
<journal-title>Frontiers in Public Health</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Public Health</abbrev-journal-title>
<issn pub-type="epub">2296-2565</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpubh.2021.642163</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Public Health</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Operational Challenges in the Use of Structured Secondary Data for Health Research</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Areco</surname> <given-names>Kelsy N.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1154231/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Konstantyner</surname> <given-names>Tulio</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/400614/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Bandiera-Paiva</surname> <given-names>Paulo</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1357212/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Balda</surname> <given-names>Rita C. X.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Costa-Nobre</surname> <given-names>Daniela T.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/173199/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Sanudo</surname> <given-names>Adriana</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Kiffer</surname> <given-names>Carlos Roberto V.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1260679/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Kawakami</surname> <given-names>Mandira D.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1313441/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Miyoshi</surname> <given-names>Milton H.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Marinonio</surname> <given-names>Ana S&#x000ED;lvia Scavacini</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Freitas</surname> <given-names>Rosa M. V.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Morais</surname> <given-names>Liliam C. C.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Teixeira</surname> <given-names>Monica L. P.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1175365/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Waldvogel</surname> <given-names>Bernadette</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Almeida</surname> <given-names>Maria Fernanda B.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/767237/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Guinsburg</surname> <given-names>Ruth</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1160720/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Escola Paulista de Medicina, Universidade Federal de S&#x000E3;o Paulo</institution>, <addr-line>S&#x000E3;o Paulo</addr-line>, <country>Brazil</country></aff>
<aff id="aff2"><sup>2</sup><institution>Funda&#x000E7;&#x000E3;o Sistema Estadual de An&#x000E1;lise de Dados</institution>, <addr-line>S&#x000E3;o Paulo</addr-line>, <country>Brazil</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Ann Borda, The University of Melbourne, Australia</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Christoph Stallmann, Otto von Guericke University Magdeburg, Germany; Everton Silva, University of Brasilia, Brazil</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Tulio Konstantyner <email>tkmed&#x00040;uol.com.br</email></corresp>
<fn fn-type="other" id="fn001"><p>This article was submitted to Digital Public Health, a section of the journal Frontiers in Public Health</p></fn></author-notes>
<pub-date pub-type="epub">
<day>15</day>
<month>06</month>
<year>2021</year>
</pub-date>
<pub-date pub-type="collection">
<year>2021</year>
</pub-date>
<volume>9</volume>
<elocation-id>642163</elocation-id>
<history>
<date date-type="received">
<day>17</day>
<month>12</month>
<year>2020</year>
</date>
<date date-type="accepted">
<day>24</day>
<month>05</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2021 Areco, Konstantyner, Bandiera-Paiva, Balda, Costa-Nobre, Sanudo, Kiffer, Kawakami, Miyoshi, Marinonio, Freitas, Morais, Teixeira, Waldvogel, Almeida and Guinsburg.</copyright-statement>
<copyright-year>2021</copyright-year>
<copyright-holder>Areco, Konstantyner, Bandiera-Paiva, Balda, Costa-Nobre, Sanudo, Kiffer, Kawakami, Miyoshi, Marinonio, Freitas, Morais, Teixeira, Waldvogel, Almeida and Guinsburg</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license> </permissions>
<abstract><p><bold>Background:</bold> In Brazil, secondary data for epidemiology are largely available. However, they are insufficiently prepared for use in research, even when it comes to structured data since they were often designed for other purposes. To date, few publications focus on the process of preparing secondary data. The present findings can help in orienting future research projects that are based on secondary data.</p>
<p><bold>Objective:</bold> Describe the steps in the process of ensuring the adequacy of a secondary data set for a specific use and to identify the challenges of this process.</p>
<p><bold>Methods:</bold> The present study is qualitative and reports methodological issues about secondary data use. The study material was comprised of 6,059,454 live births and 73,735 infant death records from 2004 to 2013 of children whose mothers resided in the State of S&#x000E3;o Paulo - Brazil. The challenges and description of the procedures to ensure data adequacy were undertaken in 6 steps: (1) problem understanding, (2) resource planning, (3) data understanding, (4) data preparation, (5) data validation and (6) data distribution. For each step, procedures, and challenges encountered, and the actions to cope with them and partial results were described. To identify the most labor-intensive tasks in this process, the steps were assessed by adding the number of procedures, challenges, and coping actions. The highest values were assumed to indicate the most critical steps.</p>
<p><bold>Results:</bold> In total, 22 procedures and 23 actions were needed to deal with the 27 challenges encountered along the process of ensuring the adequacy of the study material for the intended use. The final product was an organized database for a historical cohort study suitable for the intended use. Data understanding and data preparation were identified as the most critical steps, accounting for about 70% of the challenges observed for data using.</p>
<p><bold>Conclusion:</bold> Significant challenges were encountered in the process of ensuring the adequacy of secondary health data for research use, mainly in the data understanding and data preparation steps. The use of the described steps to approach structured secondary data and the knowledge of the potential challenges along the process may contribute to planning health research.</p></abstract>
<kwd-group>
<kwd>public health</kwd>
<kwd>datasets as topic</kwd>
<kwd>population studies in public health</kwd>
<kwd>Death Certificates</kwd>
<kwd>Birth Certificates</kwd>
<kwd>secondary health data</kwd>
</kwd-group>
<contract-sponsor id="cn001">Funda&#x000C3;&#x000A7;&#x000C3;&#x000A3;o de Amparo &#x000C3; Pesquisa do Estado de S&#x000C3;&#x000A3;o Paulo<named-content content-type="fundref-id">10.13039/501100001807</named-content></contract-sponsor>
<counts>
<fig-count count="1"/>
<table-count count="4"/>
<equation-count count="0"/>
<ref-count count="48"/>
<page-count count="11"/>
<word-count count="8882"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>Introduction</title>
<p>Secondary health data supports information production to develop and evaluate preventive and therapeutic strategies, services, programs, and health policies. It is quite advantageous to be able to use these data for research purposes since they have been already collected (<xref ref-type="bibr" rid="B1">1</xref>&#x02013;<xref ref-type="bibr" rid="B3">3</xref>).</p>
<p>In Brazil, social and health data, collected continuously or periodically, are, in general, structured (variables with previously established meaning and coding), consolidated, anonymized, and with unrestricted public access (<xref ref-type="bibr" rid="B4">4</xref>&#x02013;<xref ref-type="bibr" rid="B11">11</xref>). Along with the data, the distribution agencies also make the materials available for their understanding, such as operational manuals, dictionaries of variables, and models of collection instruments, as well as tools for their visualization in the form of graphs, tables, or maps (<xref ref-type="bibr" rid="B7">7</xref>&#x02013;<xref ref-type="bibr" rid="B11">11</xref>).</p>
<p>In the United States, for example, the Center for Disease Control and Prevention (CDC) internet site makes a lot of structured secondary health data publicly available (<xref ref-type="bibr" rid="B12">12</xref>, <xref ref-type="bibr" rid="B13">13</xref>), and also provides restricted access data for research (<xref ref-type="bibr" rid="B14">14</xref>), and tools to query the data (<xref ref-type="bibr" rid="B13">13</xref>). Data from other countries can also be found at <ext-link ext-link-type="uri" xlink:href="http://ghdx.healthdata.org/">http://ghdx.healthdata.org/</ext-link> (<xref ref-type="bibr" rid="B15">15</xref>).</p>
<p>Some of the Brazilian agencies that provide open data are the Interagency Health Information Network (RIPSA) (<xref ref-type="bibr" rid="B6">6</xref>), the Information Technology Department of the Public Health Care System (DATASUS) (<xref ref-type="bibr" rid="B7">7</xref>), the Brazilian Institute of Geography and Statistics (IBGE) (<xref ref-type="bibr" rid="B8">8</xref>), the S&#x000E3;o Paulo State Data Analysis System Foundation (SEADE) (<xref ref-type="bibr" rid="B9">9</xref>) and the Brazilian Open Data Portal (<xref ref-type="bibr" rid="B10">10</xref>, <xref ref-type="bibr" rid="B11">11</xref>).</p>
<p>Considering the whole country, the main source of secondary health data is DATASUS (<xref ref-type="bibr" rid="B7">7</xref>). Such available health data are collected through DATASUS (<xref ref-type="bibr" rid="B7">7</xref>)&#x00027;s Information Systems and stored in administrative databases. The use of secondary health data such as those from DATASUS (<xref ref-type="bibr" rid="B7">7</xref>) databases has become increasingly frequent as can be seen, searching the PubMed (MEDLINE) database under the query &#x0201C;datasus (Title/Abstract).&#x0201D;</p>
<p>Among the DATASUS&#x00027;s Information System, the Mortality Information System (SIM) and the Live Birth Information System (SINASC), stand out for their importance in the generation of vital statistics and social indicators, living conditions, and child health (<xref ref-type="bibr" rid="B7">7</xref>, <xref ref-type="bibr" rid="B16">16</xref>&#x02013;<xref ref-type="bibr" rid="B19">19</xref>). These 2 databases contain information on all live births and all deaths informed in the whole country, independently if an individual is a user of the Brazilian public health care or not.</p>
<p>Live births and death data are collected in the Live Birth Certificates (LBC) and Death Certificates (DC), respectively. The paper forms, LBC and DC, are completed in three copies: (a) feeds the national database SIM and SINASC; (b) is retained in the Civil Registry office in which the birth or death event is registered and (c) is retained in the service informing the event (<xref ref-type="bibr" rid="B9">9</xref>).</p>
<p>In the State of S&#x000E3;o Paulo (SP), SEADE Foundation is the institution responsible for the collection, organization, analysis, and dissemination of these records (death and live births) under which the vital statistics of SP are produced from data present in LBC and DC (<xref ref-type="bibr" rid="B9">9</xref>). The Civil Registry offices of 645 municipalities in SP send the completed LBC and DC forms to SEADE monthly. After being entered into SEADE&#x00027;s system, infant death data (infant under 1 year of age) is linked to birth data (<xref ref-type="bibr" rid="B9">9</xref>). The linked file consists of deaths of infants born in a given year including their birth variables. While anonymized births and death records are publicly accessible on an internet site, linked files are not.</p>
<p>A similarly linked dataset of live births and infant death who died in the United States, Puerto Rico, The Virgin Islands, and Guam are available as downloadable data files on the internet site <ext-link ext-link-type="uri" xlink:href="https://www.cdc.gov/nchs/data_access/vitalstatsonline.htm">https://www.cdc.gov/nchs/data_access/vitalstatsonline.htm</ext-link>. The linked birth and infant death dataset is also available in birth cohort data format, with the complete description of these data (<xref ref-type="bibr" rid="B20">20</xref>).</p>
<p>In scientific research, the concomitant use of these data makes it possible to calculate the risk estimates of infant death and its age components, to analyse risk factors or determinants of specific outcomes, to estimate cause-specific mortality rates, and to analyze time series and spatial and ecological studies (<xref ref-type="bibr" rid="B2">2</xref>, <xref ref-type="bibr" rid="B16">16</xref>, <xref ref-type="bibr" rid="B21">21</xref>&#x02013;<xref ref-type="bibr" rid="B23">23</xref>).</p>
<p>Despite this favorable and stimulating scenario, secondary data is not always ready for use. In these situations, there are difficulties to be considered, such as limiting the data to certain geographic areas or periods, when there are changes in the way of collecting the variables, lack of standardization in the data format, discontinuity in the collection of some data over time or variation in coverage (<xref ref-type="bibr" rid="B24">24</xref>). Besides, it may be necessary to select them according to conditions related to the inclusion or exclusion criteria, which is done by transforming data from the original database to obtain the study population. Anyway, there are many situations, conditions, or factors that can impact the usability of the data for purposes other than those for which it was collected. To answer the research questions, the data must be organized in a way that their handling is quick and easy. However, depending on the resources available, these operations may not be easily implemented.</p>
<p>The knowledge of the steps and obstacles that can arise using secondary data for specific purposes in research potentially allows to identify critical points, and, consequently, to plan actions and direct resources for the effective execution of research projects. Thus, this study aimed to describe the steps necessary to ensure an adequate set of structured secondary health data for use in quantitative research and to identify the challenges of this process. The data will be considered adequate for use if they fit the purpose of the research; and are ready to be used in the planned analyses, and there are no ethical constraints in using them.</p></sec>
<sec sec-type="materials and methods" id="s2">
<title>Materials and Methods</title>
<p>The present study is a qualitative study that reports methodological issues related to the use of a secondary health database.</p>
<p>The study material consisted of all records of live births between 2004 and 2013, and infant deaths (0&#x02013;365 days) of children born from mothers residing in one of the 645 municipalities in the State of S&#x000E3;o Paulo, totaling 6,059,454 births and 73,735 deaths. These records were originated in the Civil Registry Offices and made available in digital format by SEADE for the execution of a project on neonatal mortality. These secondary data were called &#x0201C;input data&#x0201D; (<xref ref-type="bibr" rid="B25">25</xref>&#x02013;<xref ref-type="bibr" rid="B27">27</xref>). These data were anonymized, following Brazil&#x00027;s General Data Protection Law, which has respect for privacy as its basic principle (<xref ref-type="bibr" rid="B28">28</xref>).</p>
<p>This study is part of a project on neonatal mortality carried out at the Federal University of S&#x000E3;o Paulo and was approved by the Research Ethics Committee of the institution under opinion 2.580.929 of 08/04/2018. The referred project is entitled &#x0201C;Secular trend, spatial evolution and maternal and neonatal conditions associated with early and late neonatal mortality due to respiratory disorders, infections, congenital anomalies and perinatal asphyxia in the state of S&#x000E3;o Paulo between 2002-2015.&#x0201D;</p>
<p>Data should be prepared for a cohort study in which live births would be classified into two groups: those who died between 0 and 27 days and those who were alive until the 27th day of life. For deaths, data come from SEADE Foundation&#x00027;s database of &#x0201C;Death linked to birth.&#x0201D; The linked file consists of death records of infants born in a given year including their birth variables. All planned analyses would be made according to the cause of death and age of death (1st hour, 1st 24 h, 0&#x02013;6 days, and 7&#x02013;27 days after birth).</p>
<p>In this study, the adequacy of secondary data for use in research refers to the potential of the data set to meet planned analysis needs, that is, the dataset is ready for starting data analysis. In this context, the approach of the present study was to divide the data adequacy process into steps, following their function.</p>
<p>Initially, the data adequacy process was based on the first three steps of the CRISP-DM (Cross Industry Standard Process for Data Mining) data science technique (<xref ref-type="bibr" rid="B29">29</xref>, <xref ref-type="bibr" rid="B30">30</xref>), which precede the data analysis: problem understanding, data understanding, and data preparation. Also, it was necessary to include three other steps to organize other procedures and operational challenges that went beyond this scope. Thereby, the challenges and description of the procedures to ensure data adequacy were undertaken in 6 steps: (step 1) problem understanding, aimed at understanding the use of data and surveying its characteristics; (step 2) resource planning, aimed at human resources, hardware, and software for subsequent steps; (step 3) data understanding, aimed at collecting and understanding the meaning and organization of the input data; (step 4) data preparation, intended for handling input data and making the output data set; (step 5) data validation, intended for data homologation; and (step 6) data distribution, for the delivery of approved output data, prepared for the specific use and ready for handling. The sequence of steps is presented as a non-cyclical path since moving back to previous steps is not expected (<xref ref-type="fig" rid="F1">Figure 1</xref>).</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Steps toward ensuring the adequacy of structured secondary health data for use in quantitative research.</p></caption>
<graphic xlink:href="fpubh-09-642163-g0001.tif"/>
</fig>
<p>For the execution of all steps, periodic meetings were planned with the participation of 12 researchers, users of the final data, who formed a working group composed of professionals in the areas of healthcare (doctors and physiotherapists) and Formal Sciences (computing and statistics).</p>
<p>The main research questions are sufficiently defined above. In summary, the premises of this study are: secondary data are not ready for the intended use; there is evidence of the relevance of these data for quantitative research purposes; and the adequacy of the data cannot be determined effectively with the use of simple tools which are commonly used for data preparation.</p>
<p>Based on these premises and the established steps, procedures were planned to ensure the adequacy of the study material for the intended use, and as defined by the working group.</p>
<p>In the execution of the procedures, the operational challenges encountered and the actions to face them were observed and recorded. The existence of any barrier to the use of data was considered an &#x0201C;operational challenge.&#x0201D;</p>
<p>For each step, procedures, challenges encountered, actions to cope with them, and partial results were described. To identify the most labor-intensive tasks in this process, the steps were assessed by adding the number of procedures, challenges, and coping actions. The highest values were assumed to indicate the most critical steps.</p></sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<p><xref ref-type="table" rid="T1">Table 1</xref> presents the 22 procedures distributed in the 6 steps of the adequacy process of structured secondary health data for specific uses in quantitative research. The steps of problem understanding, data understanding, and data preparation stood out concerning the quantity with 5, 7, and 6 procedures, respectively.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Procedures at each step for ensuring the adequacy of structured secondary health data for specific use in quantitative research.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Step</bold></th>
<th valign="top" align="left"><bold>Procedures</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1. Problem understanding</td>
<td valign="top" align="left">1. Assessment of the characteristics of secondary input data</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x02022; Content: what the data represents in the real world, source of data, the context in which it was collected</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x02022; Estimated volume: number of records and size of expected files</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x02022; Expected data file format</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">2. Assessment of the characteristics of the research</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x02022; Population and period under study</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x02219;Inclusion and exclusion criteria for selection</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x02022; Study design and analysis unit</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x02022; Variables involved in the main research questions, objectives, and hypotheses</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">3. Assessment of the characteristics of the output data</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x02022; Estimated output data volume: number of records or file size</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x02022; The desired format for delivery of output data</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">4. Checking the availability of input data and variable dictionaries</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">5. Evaluation of the ethical aspects and technical feasibility of data adequacy for the research</td>
</tr>
<tr>
<td valign="top" align="left">2. Resource planning</td>
<td valign="top" align="left">6. Sizing up human resources</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">7. Sizing up computational resources (hardware and software platform)</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x02022; Volume and format of input data</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x02022; Support for the operations required to adjust the input data</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x02022; Estimated volume and format of output data</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x02022; Performance and data volume limits for eligible computing resources</td>
</tr>
<tr>
<td valign="top" align="left">3. Data understanding</td>
<td valign="top" align="left">8. Obtaining secondary data files and variable dictionaries</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">9. Understanding the variable dictionaries related to the input data and creating the research variables dictionary for each type of file</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">10. Inventory of data files: name and extension, size in bytes, and number of records</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">11. Assessment of the existence of a unique record identifier (primary key) in each data file</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">12. Inventory of the variables contained in the data files: name, type, and size</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">13. Exploratory data analysis for completeness</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">14. Elaboration of the data extraction plan for the research</td>
</tr>
<tr>
<td valign="top" align="left">4. Data preparation</td>
<td valign="top" align="left">15. Execution of the data extraction plan</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">16. Exploratory data analysis to detect invalid content and assess the homogeneity in a data filling</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">17. Data cleaning and transformation to generate research variables</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">18. Updating the search variable dictionary</td>
</tr>
<tr>
<td valign="top" align="left">5. Data validation</td>
<td valign="top" align="left">19. Exploratory analysis of the transformed data for comparison with the original data</td>
</tr>
<tr>
<td valign="top" align="left">6. Data distribution</td>
<td valign="top" align="left">20. Exporting the database to the specified format (s)</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">21. Reduction of the database to contain only the research variables</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">22. Distribution of the database and dictionary of research variables</td>
</tr>
</tbody>
</table>
</table-wrap>
<p><xref ref-type="table" rid="T2">Table 2</xref> presents the operational challenges encountered and the actions to face them. A total of 27 operational challenges were identified, of which 66.7% (18 from 27) were from the steps of data understanding and data preparation (steps 3 and 4), corresponding to operational challenges 6&#x02013;23. In these steps were also concentrated most actions to face these challenges (15 from 23), coping actions 5&#x02013;19.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Operational challenges identified in the steps for ensuring the adequacy of structured secondary health data for specific use in quantitative research.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Step</bold></th>
<th valign="top" align="left"><bold>Operational challenges</bold></th>
<th valign="top" align="left"><bold>Coping actions</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1. Problem understanding</td>
<td valign="top" align="left">1. Unavailability of the complete set of data files for immediate access<break/>2. Lack of definition on how to access the variable dictionary<break/>3. Interdisciplinary communication in the team<break/>4. Establishment of consensus in the decisions and definitions</td>
<td valign="top" align="left">1. Meetings with the institution providing the data<break/>2. Recording of decisions and definitions<break/>3. Obtaining a sample of the data to assess the technical feasibility of data adequacy for the specific use</td>
</tr>
<tr>
<td valign="top" align="left">2. Resource planning</td>
<td valign="top" align="left">5. Need to optimize cost and preparation time for a large volume of data (&#x0007E;10 Gigabytes) in more than one format</td>
<td valign="top" align="left">4. Prioritizing the use of available human and computational resources and planning the acquisition of complementary computational resources to minimize the training time for human resources</td>
</tr>
<tr>
<td valign="top" align="left">3. Data understanding</td>
<td valign="top" align="left">6. Need to improve understanding of variables<break/> 7. Variables that have changed their format over time<break/> 8. Multiple files with different structures<break/> 9. No unique identifiers of records<break/>10.File structure differ from data dictionary description<break/>11. Variables filled with codes from other information systems</td>
<td valign="top" align="left">5. Consultation with other sources of information and exchange of information in periodic meetings<break/> 6. Storage of data in database tables, using text fields<break/> 7. Unique identifier insertion of records to make them logically accessible<break/> 8. Log and reuse of commands (queries) in SQL language when possible<break/> 9. Making variable dictionaries with standardized names<break/>10. Elaboration of the data extraction and combination plan: reduction of the number of tables; standardization of data structures; adding the source in the primary key of the tables; and identification of variables to filter the records of interest</td>
</tr>
<tr>
<td valign="top" align="left">4. Data preparation</td>
<td valign="top" align="left">12. Multiple tables<break/>13. Multiple values to denote Null content<break/>14. Different filling formats in date variables<break/>15. Invalid values<break/>16. Different filling formats in numeric variables<break/>17. Variables with mixed content<break/>18. Variables with multiple contents<break/>19. Duplicates in variables with multiple contents<break/>20. No rules for cross consistency of related variables<break/>21. Variable filled with code dependent on an external database<break/>22. No direct reference to the external databases used<break/>23. No single variable for data file integration</td>
<td valign="top" align="left">11. Reducing the original data (multiple tables) to two tables (table union)<break/>12. Extraction of records of interest after combining data<break/>13. Standardization of null content and recount of nulls<break/>14. Elaboration and execution of the cross-consistency rules of the variables<break/>15. Standardization of variable formats<break/>16. Search for official databases to decode variables dependent on external codes<break/>17. Incorporation of the description of external codes in the research database<break/>18. Separation of variables with multiple contents into new variables for decomposition into single content<break/>19. Data integration using a set of variables common to the tables</td>
</tr>
<tr>
<td valign="top" align="left">5. Data validation</td>
<td valign="top" align="left">24. No single report with the same scope in the original data source for comparison with prepared data</td>
<td valign="top" align="left">20. Validation of the transformed data based on the expected data volume and the frequency distribution of each variable according to a time dimension</td>
</tr>
<tr>
<td valign="top" align="left">6. Data distribution</td>
<td valign="top" align="left">25. Big data volume (approximately 10 Gigabytes)<break/>26. Need to deliver output data in more than one format<break/>27. Need for storage and backup of work files</td>
<td valign="top" align="left">21. Use of a statistical package to incorporate the dictionary of variables into the data<break/>22. Use of converter software to export data and variable dictionary<break/>23. Creation of private cloud for data distribution and users with different access levels</td>
</tr>
</tbody>
</table>
</table-wrap>
<p><xref ref-type="table" rid="T3">Table 3</xref> shows the results achieved after the execution of the procedures and actions to face the operational challenges corresponding to each step. In one case due to the need to create a new variable not previously defined, it was necessary to return to the data preparation step.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Results achieved at each step for ensuring the adequacy of structured secondary health data for specific use in quantitative research.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Step</bold></th>
<th valign="top" align="left"><bold>Results achieved</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1. Problem understanding</td>
<td valign="top" align="left"><italic>Definitions and Information acquired</italic></td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Input data: annual records of infant deaths (&#x0007E;8 thousand) and live births (&#x0007E;600 thousand) in Microsoft&#x000AE; Excel spreadsheet format (<italic>.xlsx</italic>) (<xref ref-type="bibr" rid="B41">41</xref>); deaths deterministically linked to births (<xref ref-type="bibr" rid="B25">25</xref>, <xref ref-type="bibr" rid="B26">26</xref>), with coded diagnoses (<xref ref-type="bibr" rid="B42">42</xref>); and variable dictionaries available (<xref ref-type="bibr" rid="B43">43</xref>)</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Output data: organized for the cohort study with neonatal death as the main outcome (&#x0007E;70% of infant deaths); suitable for the selection of specific causes of death; with group identification and classification of cases of congenital anomaly in death or live birth</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Distribution file format: flat-file (.<italic>csv</italic>) and variable dictionaries in Microsoft&#x000AE; Word 2010 (<xref ref-type="bibr" rid="B41">41</xref>) format (<italic>.docx</italic>); data and variable dictionary embedded in data file format of the statistical packages SPSS v24&#x000AE; (<xref ref-type="bibr" rid="B44">44</xref>) <italic>(.sav)</italic> and Stata v15&#x000AE; (<xref ref-type="bibr" rid="B45">45</xref>) (<italic>.dta</italic>)</td>
</tr>
<tr>
<td valign="top" align="left">2. Resource planning</td>
<td valign="top" align="left"><italic>For steps 3, 4 and 5</italic></td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Hardware: portable computer with 16 gigabytes of RAM (random access memory) and 1 terabyte hard disk</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Operating System: Microsoft&#x000AE; Windows 10&#x000AE; (<xref ref-type="bibr" rid="B46">46</xref>)</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Software: Microsoft&#x000AE; Office Professional 2010&#x000AE; (<xref ref-type="bibr" rid="B41">41</xref>), Microsoft&#x000AE; SQL Server 2012 Express&#x000AE; (<xref ref-type="bibr" rid="B34">34</xref>), SPSS v24&#x000AE; (<xref ref-type="bibr" rid="B44">44</xref>) and Stat Transfer v14&#x000AE; (<xref ref-type="bibr" rid="B47">47</xref>)</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Human Resources <italic>(Peopleware)</italic>: training in SQL language</td>
</tr>
<tr>
<td/>
<td valign="top" align="left"><italic>For step 6</italic></td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Platforms: Cloud storage system (OwnCloud version 10.5.0), free and open source running on a virtual machine based on the Linux operating system (Fedora Server 31, kernel 5.7.15), maintained as a Virtual Machine in the Research Datacenter of the Federal University of S&#x000E3;o Paulo (<ext-link ext-link-type="uri" xlink:href="https://www.dis.epm.br/&#x00023;/parque_maquinas">https://www.dis.epm.br/&#x00023;/parque_maquinas</ext-link>)</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Peopleware: Network and infrastructure analyst</td>
</tr>
<tr>
<td valign="top" align="left">3. Data understanding</td>
<td valign="top" align="left"><italic>Stored, understood and identified data</italic></td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Anonymized annual data has been imported into 20 tables</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Infant death files contained variables present in the Death and Live Birth Certificates</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; The live birth files contained variables from Live Birth Certificates</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; The causes of death were reported in 6 variables (basic cause, line A, line B, line C, line D, line II) and the lines could contain one or more International Classification of Diseases codes, 10th Revision (<xref ref-type="bibr" rid="B42">42</xref>)</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Information on anomalies was present in the records of Death and Live Birth Certificates</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Sequential identification has been added as a primary key in the tables</td>
</tr>
<tr>
<td/>
<td valign="top" align="left"><italic>Data files extraction and merging plan</italic></td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Data Merging (Combination of input data) to reduce tables and include data source identification for the primary key composition</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Data Extraction based on two variables: age and municipality of residence</td>
</tr>
<tr>
<td valign="top" align="left">4. Data preparation</td>
<td valign="top" align="left"><italic>Combined and extracted data</italic></td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Data reduced to 2 tables with defined primary key and unified variables formatted as text type</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; 50,842 neonatal deaths were extracted, without changing the number of live birth records</td>
</tr>
<tr>
<td/>
<td valign="top" align="left"><italic>Transformed and integrated data</italic></td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Standardization and cleaning of data, observing the presence of invalid values; invalid formats of data type variables; variables filled in as number and text; possibility to correct the format; errors revealed by crossing related variables</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Integration of the two tables for the cohort study using the common variables: 50,247 neonatal death records were identified among the Live Birth Certificates</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; The causes of death were stored in 27 new variables and the duplicates were eliminated</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Cases of congenital anomaly have been identified; diagnoses of anomaly in live births were organized into 10 new variables; deaths with anomaly were classified into 11 groups (present or absent)</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; The descriptions of the external codes were incorporated into the data: diagnoses (<xref ref-type="bibr" rid="B42">42</xref>) and the municipalities (IBGE) (<xref ref-type="bibr" rid="B48">48</xref>)</td>
</tr>
<tr>
<td valign="top" align="left">5. Data validation</td>
<td valign="top" align="left"><italic>Validated data and the completed dictionary of variables</italic></td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Counting the total number of neonatal deaths and exploring the main variables per year</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; The need to create a new variable was identified, going back to the previous step</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Database was approved and the dictionary of variables was finalized</td>
</tr>
<tr>
<td valign="top" align="left">6. Data distribution</td>
<td valign="top" align="left"><italic>Data arranged in the specified and distributed formats</italic></td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Data exported from MS&#x000AE; SQLServer&#x000AE; (<xref ref-type="bibr" rid="B34">34</xref>) in .csv format and imported in the SPSS&#x000AE; (<xref ref-type="bibr" rid="B44">44</xref>) statistical package (.sav)</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; The description of the variables and their values has been incorporated into the data file (.sav)</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; .sav file has been converted to the Stata&#x000AE; statistical package data file format (.dta) (<xref ref-type="bibr" rid="B45">45</xref>, <xref ref-type="bibr" rid="B47">47</xref>)</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x000A0;&#x02022; Cloud storage (<ext-link ext-link-type="uri" xlink:href="https://doc.bioinfo.unifesp.br/cloud">https://doc.bioinfo.unifesp.br/cloud</ext-link>) for sharing or distribution</td>
</tr>
</tbody>
</table>
</table-wrap>
<p><xref ref-type="table" rid="T4">Table 4</xref> summarizes the number of procedures (detailed in <xref ref-type="table" rid="T1">Table 1</xref>), operational challenges (detailed in <xref ref-type="table" rid="T2">Table 2</xref>), coping actions (detailed in <xref ref-type="table" rid="T2">Table 2</xref>), and the criticality ranking for each step. Starting with the most critical step, the resulting ranking was: (1st) data preparation; (2nd) data understanding; (3rd) problem understanding; (4th) data distribution; (5th) resource planning; and (6th) data validation.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Number of procedures<xref ref-type="table-fn" rid="TN1"><sup>&#x0002A;</sup></xref>, operational challenges<xref ref-type="table-fn" rid="TN2"><sup>&#x0002A;&#x0002A;</sup></xref>, coping actions<xref ref-type="table-fn" rid="TN2"><sup>&#x0002A;&#x0002A;</sup></xref> and criticality ranking of the steps for ensuring an adequate set of structured secondary data in the health field for specific use in quantitative research.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Step</bold></th>
<th valign="top" align="center"><bold>Procedures</bold></th>
<th valign="top" align="center"><bold>Operational challenges</bold></th>
<th valign="top" align="center"><bold>Coping actions</bold></th>
<th valign="top" align="center"><bold>Critical order</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1. Problem understanding</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">3</td>
</tr>
<tr>
<td valign="top" align="left">2. Resource planning</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">5</td>
</tr>
<tr>
<td valign="top" align="left">3. Data understanding</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">2</td>
</tr>
<tr>
<td valign="top" align="left">4. Data preparation</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">12</td>
<td valign="top" align="center">9</td>
<td valign="top" align="center">1</td>
</tr>
<tr>
<td valign="top" align="left">5. Data validation</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">6</td>
</tr>
<tr style="border-bottom: thin solid #000000;">
<td valign="top" align="left">6. Data distribution</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">4</td>
</tr> <tr>
<td valign="top" align="left">Total</td>
<td valign="top" align="center">22</td>
<td valign="top" align="center">27</td>
<td valign="top" align="center">23</td>
<td valign="top" align="center">&#x02013;</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="TN1"><label>&#x0002A;</label><p><italic>The procedures are listed in <xref ref-type="table" rid="T1">Table 1</xref></italic>,</p></fn>
<fn id="TN2"><label>&#x0002A;&#x0002A;</label><p><italic>operational challenges and coping actions in <xref ref-type="table" rid="T2">Table 2</xref></italic>.</p></fn>
</table-wrap-foot>
</table-wrap>
<p>The size of files and number of records processed are presented in <xref ref-type="supplementary-material" rid="SM1">Supplementary Material 1</xref>. The data description of the final dataset is presented in <xref ref-type="supplementary-material" rid="SM2">Supplementary Material 2</xref>.</p>
<p>In the research project for which the data were prepared, the objectives, study design, and the intended use of the data were clearly defined, but they had not been sufficiently detailed for the elaboration of the research database, a task that was done only after receiving the data or a sample of it, in step 1 (<xref ref-type="table" rid="T1">Tables 1</xref>&#x02013;<xref ref-type="table" rid="T3">3</xref>). The tasks in step 1 help to refine the data usage needs that were often not sufficiently clear in the research project or that have changed. For example, in the present study, initially, to classify a death with congenital anomaly, it was expected to involve only checking if one of the causes of death was in the code range (Congenital anomalies: Q0&#x02013;Q99). Reviewing the needs of the project, it was realized that it was also necessary to classify individuals into groups of anomalies, being that the same individual could present anomalies of one or more groups (<xref ref-type="table" rid="T3">Table 3</xref>, <xref ref-type="supplementary-material" rid="SM2">Supplementary Material 2</xref>).</p>
<p>After performing the procedures in step 1- <xref ref-type="table" rid="T1">Table 1</xref>, it was possible to observe that the data were not prepared in an adequate way for the intended use. For example, the input data were distributed in several files with different structures. Therefore, in order to represent the cohort design study in a flat-file, the birth data should be combined with the &#x0201C;death linked to birth&#x0201D; data. It was also necessary to classify all live births in two groups, according to neonatal death outcome, as declared on the referred research project.</p>
<p>The procedures showed in <xref ref-type="table" rid="T1">Table 1</xref> and the coping actions of the challenges listed in <xref ref-type="table" rid="T2">Table 2</xref> were necessary tasks to prepare these input data for the intended use.</p>
<p>The execution of the procedures and actions addressing the challenges in each step resulted in a database which was ensured for adequacy in its intended use, according to the defined requirements in step 1. According to the definitions regarding the contents and file formats of the output data, the final product of the described process was a simple file, distributed in 3 formats, containing the consolidated data and integrated to represent the cohort of live births between 2004 and 2013 of children born from mothers residing in the State of S&#x000E3;o Paulo, having neonatal death (0&#x02013;27 days) as the main outcome; the cases of congenital anomalies identified and classified in groups; and the causes of death, separated, and without any duplication; and the other variables of interest collected and available in the Live Birth or Death Certificates (<xref ref-type="table" rid="T3">Table 3</xref>).</p></sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>The present study described the steps for ensuring the adequacy of secondary data on live births and infant deaths for use in research on neonatal mortality and identified 27 operational challenges of this process. To accomplish this task, 22 procedures and 23 actions were needed to face the challenges encountered, organized in 6 steps. The steps of data preparation and data understanding were identified as the most labor-intensive tasks.</p>
<p>These two steps are those that demand human resources with specific skills and appropriate computational resources to support operations for manipulating databases. In fact, knowledge of the health care system, the background and processes involved in the creation of the data, an understanding of the content, and the ability to think in terms of data structures are essential skills to deal with the data preparation and challenges almost always encountered in the procedures described in the study. And, above all, sufficient human resources must be available for working with these data for the reasons mentioned. Coeli et al. proposed to address the following issues in the training of human resources to work with secondary data: &#x0201C;SQL (Structured Query Language), linking of records, integration of unstructured data, data mining and computational modeling of complex systems&#x0201D; (<xref ref-type="bibr" rid="B31">31</xref>).</p>
<p>According to a Brazilian study that proposed to create the National Health Database Centered on the individual, using administrative and epidemiological databases (2000&#x02013;2015) from four DATASUS Information Systems, cleaning and standardizing data &#x0201C;are relevant and laborious tasks, given the high frequency of inconsistent, incomplete or misspelled data&#x0201D; (<xref ref-type="bibr" rid="B3">3</xref>). Our results corroborate the findings of this study, once cleaning and standardizing data are data preparation tasks (step 4), the most labor-intensive step assessed by the present study (<xref ref-type="bibr" rid="B3">3</xref>).</p>
<p>No problems were encountered that would result in unsuitable data for the defined use. However, challenges faced at any step may impact the research results. Potential challenges at each step are further outlined and discussed below.</p>
<p>Initially, in step 1 &#x0201C;problem understanding,&#x0201D; the periodic meetings of the working group made it possible to align the technical language between professionals from different areas, minimizing possible communication failures and, consequently, facilitating consensual decision-making. The lack of clarity in defining the needs for using the data can lead to rework, especially in the data preparation step, impacting the time and cost to carry out the research project, in addition to the inadequate sizing up of resources (<xref ref-type="bibr" rid="B2">2</xref>). This reinforces the importance of integrating the group to reach the final product, that is, a suitable and simple database file. The success in executing the procedures and actions to face the operational challenges at this step is also due to the availability of data at the beginning of the process, which made it possible to analyze the technical feasibility of the preparation of a database suitable for the intended use and opened a more sustained path for the execution of the next steps. Ethical issues must always permeate the use of secondary health data: every individual has the right to confidentiality, secrecy, and privacy of personal health information, regardless of the medium in which it circulates. Thus, in steps 1 and 3, the recognition of these issues is essential to avoid the misuse of personal information in research (<xref ref-type="bibr" rid="B32">32</xref>).</p>
<p>In step 2 &#x0201C;resource planning,&#x0201D; the hardware and software resources existing in the institution proposing the research project were first allocated, avoiding unnecessary expenses. As available human resources, only members of the research group were considered. The choice of software was based on the characteristics of the data sources; the estimates of the volume of input and output data; the data delivery format for the intended use; the functionalities of the tools for data processing and delivery; the budget limits of the research project; and the researchers&#x00027; experience with the use of resources, aiming to minimize the execution time for data adequacy and avoid errors in the entire process.</p>
<p>Step 3 &#x0201C;data understanding&#x0201D; started with obtaining the secondary data files for the research institution and ended with the data extraction plan. The use of a Relational Database Management Systems (RDBMS) enabled the exploration of around 6 million records distributed in multiple files. Despite a large amount of data, the commands in SQL were executed with good performance in the specified equipment with 16 Gb of RAM, in an interactive way to visualize the data (<xref ref-type="bibr" rid="B33">33</xref>). Besides exploring the data content and filling patterns, and consulting the variable dictionaries and operational manuals, the discussions in the multidisciplinary working group facilitated the understanding of the data. Often, the major impediment to the use of secondary data is the unavailability of dictionaries of variables or difficulty in accessing them, which makes the task of understanding these data very arduous. Thus, the researcher&#x00027;s unclarified questions about the data can impair its use. Data cannot be used without an understanding of what it represents and how it was generated, as this can result in errors of interpretation, impacting the results of the research. So, those responsible for the custody of databases must make abundant documentation about their data (<xref ref-type="bibr" rid="B31">31</xref>). Moreover, the lack of variability and incomplete filling of variables can make the dataset unfeasible.</p>
<p>RDBMS are suitable for operation with structured data. For steps 3 and 4, an Open Source alternative for MS&#x000AE; SQLServer&#x000AE; is MySQL (<xref ref-type="bibr" rid="B33">33</xref>, <xref ref-type="bibr" rid="B34">34</xref>). As an alternative to SQL embedded in any RDBMS, there are also other programming languages, such as Python (<xref ref-type="bibr" rid="B35">35</xref>), data extraction, transformation and loading tools, and statistical packages such as R (<xref ref-type="bibr" rid="B36">36</xref>) or spreadsheets. The programming language Python (<xref ref-type="bibr" rid="B35">35</xref>) or the statistical package R (<xref ref-type="bibr" rid="B36">36</xref>) can also be considered as resources for both data preparation and analysis. Spreadsheets are simpler tools and help to manipulate the data but may not support large volumes of them. Whichever tool is chosen for the treatment of the data, it is important to observe its limits regarding the number of records in the rows and variables in the columns; the ease in carrying out the joining, union, and grouping operations; and the performance of the software and the hardware. In short, the most important is that the software should be used according to the intended goal. RDBMS are very useful to prepare data but very inconvenient for statistical analysis. Nevertheless, the more analysis-oriented programs are quite capable of working with relational data structures.</p>
<p>Step 4 was the most critical for the use of the secondary data. Despite this, all the procedures and actions to face the challenges performed in this step depended on the decisions and information obtained in step 1, the resources planned in step 2, and the understanding of the data made in step 3. Prepare the data without prior understanding of them, and the needs of their use can introduce errors that will be propagated to the data analysis phase, generating invalid results. In this step, it is important to check if the transformations were successful, comparing or crossing the original values with the transformed ones, because, when a valid transformation rule to correct a format is identified, there may still be exceptions.</p>
<p>The adequacy of the data was limited to the questions provided in step 1. It is worth mentioning that the planned data transformations for one data set may not be valid for another. New challenges require other coping actions. It is likely that, in the analysis process for which the data was prepared, new unforeseen issues may arise and, therefore, new needs for organization or transformation of an existing database. Although it is not possible to foresee these future needs, the treatment of the data up to this point has allowed them to be understood, cleaned, standardized, consolidated, and reorganized in a structure that facilitates future data transformations, if necessary.</p>
<p>In step 5 &#x0201C;data validation,&#x0201D; the data were subjected to a critical assessment about the content and format of the variables. The data on live births were compared with the annual totals presented on the website of the supplier, the SEADE Foundation, considering the same definition of the study population: residents in the State of S&#x000E3;o Paulo (<xref ref-type="bibr" rid="B26">26</xref>, <xref ref-type="bibr" rid="B30">30</xref>). The similarity between the annual distributions was verified using variables of live births and neonatal deaths. Errors detected in this step may imply the need to return to the steps of understanding and preparing data. In the absence of a reliable source for data validation, it is suggested that the distribution patterns of the variable be observed according to a dimension of time and space and possible outliers and discrepancies are identified.</p>
<p>In step 6 &#x0201C;data distribution,&#x0201D; concerning the output data distribution formats, &#x0201C;<italic>.csv&#x0201D;</italic> (values separated by commas) stands out, due to the possibility of being imported to other platforms and statistical packages, ensuring data portability. In the other formats, &#x0201C;<italic>.sav&#x0201D;</italic> and &#x0201C;<italic>.dat,&#x0201D;</italic> the data and the dictionary of variables are arranged in the same file, preventing the documentation from being lost (<xref ref-type="bibr" rid="B37">37</xref>). The other file formats are a proprietary format, developed and maintained as part of each statistical software application, thus there is no guarantee that they can be read by other software. The software manuals must be consulted to import and export files saved in different formats.</p>
<p>The choice of the <italic>OwnCloud</italic> data distribution platform avoided additional costs and ensured data availability for the entire team. After distribution to the group of researchers, the research database was considered ready for use. In this case, the evaluation was made by the same group that participated in the specifications defined in step 1.</p>
<p>Finally, at the end of the data adequacy process, the data were considered understood, integrated, standardized, prepared to meet the requirements, validated, and accessible to researchers on the platform and in the specified formats.</p>
<p>The procedures of this study, described abstractly and chronologically ordered, corresponded to the planned actions and guided the execution of the process. The &#x0201C;operational challenges&#x0201D; and &#x0201C;coping actions,&#x0201D; also described abstractly, corresponded to the unforeseen obstacles and the strategies for their solution, aiming at the fulfillment of the procedures. The result of each step was described in a pragmatic way to reduce the gap between the abstract and theoretical, and the concrete and practical elements. Similar to the checklists or &#x0201C;scientific writing guides,&#x0201D; this work is potentially a proposal for guidance on research work in the health field based on secondary data because it elucidates the challenges in the steps before data analysis and addresses issues that can contribute to the orientation of researchers (<xref ref-type="bibr" rid="B38">38</xref>).</p>
<p>As the main limitation of this study, it can be considered that other data sets may present additional challenges that have not been identified. Thus, all results are limited to the data set used as material and the purpose of that research, for which the data were prepared. For example, more complex challenges can be identified when linking data requires more complex procedures than those faced in this study. On the other side, the challenges encountered in the data adequacy process may be overestimated under the following conditions: when researchers are experienced using data from a given source; when the data maintain the same structure over time; or there is no need to prepare or transform data to answer research questions.</p>
<p>Other study limitations are the challenges in classifying the level of difficulty of a data task and similarly, the time required to undertake a task was not measured. Time is related to cost, and both time and cost can be an obstacle to use the secondary data. Generally, the time depends on the data quality, and the complexity of the data preparation tasks, and the experience of the individuals who handle the data. The knowledge of the time to perform the tasks can help in planning the research.</p>
<p>Regarding the generability of the proposed steps for other secondary data structured in Brazil and other countries, we consider that, although the data appear to be suitable for use, there are two essential steps in this process: step 1 (problem understanding), aiming to understand the use of data and survey of its characteristics; and step 3 (data understanding), in order to collect and understand the meaning and how they are organized. In step 1, there are two main items to observe: the first refers to the potential of secondary data to answer the referred research questions; and the second, to ethical restrictions on the intended use. Step 3 will be necessary, except if the researcher is already completely familiar with the data and there has been no change in the structure or domain of the values of the variables. Data understanding avoids the production of information of bad quality. With regard to Brazilian data, we believe that steps 2, 4, and 5 are also applicable unless data pre-processing has already been done, is maintained by a suitable institution, and there is a data administrator available for consult. For example, combining individual health data from more than one source will not be a simple task, as there is no guarantee that individual health data collected from different systems will share unique key identifiers for individuals (<xref ref-type="bibr" rid="B39">39</xref>). In addition, many data may have inconsistencies and other quality problems, as previously mentioned (<xref ref-type="bibr" rid="B3">3</xref>). Step 6 is necessary to support a group of researchers who share the same data. In this sense, a framework for the evaluation of secondary data for use in epidemiological research can also help to decide on secondary data use. The framework proposed by Sorensen et al. includes the following items: &#x0201C;(<xref ref-type="bibr" rid="B1">1</xref>) integrity of the record of individuals, (<xref ref-type="bibr" rid="B2">2</xref>) the accuracy and degree of integrity of the recorded data, (<xref ref-type="bibr" rid="B3">3</xref>) the size of the data source, (<xref ref-type="bibr" rid="B4">4</xref>) the registration period; (<xref ref-type="bibr" rid="B5">5</xref>) accessibility, availability, and cost of data; (<xref ref-type="bibr" rid="B6">6</xref>) data format; and (<xref ref-type="bibr" rid="B7">7</xref>) possibilities of linking with other data sources (linking of records)&#x0201D; (<xref ref-type="bibr" rid="B2">2</xref>).</p>
<p>Secondary health data are relevant material to answer public health questions. Initiatives to provide secondary data previously prepared for general purposes, consolidated and clean, minimize the work of researchers in their preparation for specific purposes, and encourage the use of the data. The data from historical series of health events made available on the Data Science Platform applied to Health (PCDaS) of the Oswaldo Cruz Foundation and other epidemiological data warehouse projects are examples of these initiatives (<xref ref-type="bibr" rid="B40">40</xref>).</p>
<p>Non-structured health data are outside the scope of this study, however, challenges and tools to manipulate and integrate them with structured data can be addressed in future studies.</p></sec>
<sec sec-type="conclusions" id="s5">
<title>Conclusions</title>
<p>In summary, this study described the steps and identified the challenges in ensuring the adequacy of structured secondary health data for specific use in quantitative scientific research. Significant challenges were encountered in this process, mainly in the data understanding and data preparation steps. Both steps require specific abilities to deal with the dataset. Nonetheless, all the steps were important to get to the final product, that is, a suitable and simple database file. The results obtained suggest that any need for adjustment due to reorganization, cleaning and correction, creation or transformation of variables, change in the original format, integration with other data, or even the understanding of its content can represent an obstacle to the use of secondary data. The use of the described steps to approach structured secondary data and the knowledge of the potential challenges along the process may contribute to planning health research.</p></sec>
<sec sec-type="data-availability-statement" id="s6">
<title>Data Availability Statement</title>
<p>The data analyzed in this study is subject to the following licenses/restrictions: the data are only partially available for public access. The datasets were provided by SEADE (S&#x000E3;o Paulo State Data Analysis System Foundation). Requests to access these datasets should be directed to <ext-link ext-link-type="uri" xlink:href="http://produtos.seade.gov.br/produtos/mrc/">http://produtos.seade.gov.br/produtos/mrc/</ext-link>.</p></sec>
<sec id="s7">
<title>Ethics Statement</title>
<p>The present study is part of a project on neonatal mortality carried out at Federal University of S&#x000E3;o Paulo and was approved by the Research Ethics Committee of the institution under opinion 2.580.929 of 08/04/2018.</p></sec>
<sec id="s8">
<title>Author Contributions</title>
<p>KA, TK, MA, and RG designed the study. KA, PB-P, DC-N, RB, AS, and MK contributed for the execution of the research. KA and TK produced the draft text. KA wrote the full version of the text. All authors revised the paper and approved the final version.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p></sec>
</body>
<back>
<ack><p>The authors are grateful to Funda&#x000E7;&#x000E3;o SEADE due to the agreements between Funda&#x000E7;&#x000E3;o SEADE and Universidade Federal de S&#x000E3;o Paulo (Numbers: &#x00023;23089.004297/2008-11 and &#x00023;23089.000057/2014-95).</p>
</ack>
<sec sec-type="supplementary-material" id="s9">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpubh.2021.642163/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpubh.2021.642163/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Table_1.DOCX" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_2.DOCX" id="SM2" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/></sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>de Drumond</surname> <given-names>E F</given-names></name> <name><surname>Machado</surname> <given-names>CJ</given-names></name> <name><surname>do</surname> <given-names>Vasconcelos MR</given-names></name> <name><surname>Fran&#x000E7;a</surname> <given-names>E</given-names></name></person-group>. <article-title>Utiliza&#x000E7;&#x000E3;o de dados secund&#x000E1;rios do SIM, SINASC e SIH na produ&#x000E7;&#x000E3;o cient&#x000ED;fica brasileira de 1990 a 2006</article-title>. <source>Rev Bras Estud Popul.</source> (<year>2009</year>) <volume>26</volume>:<fpage>7</fpage>&#x02013;<lpage>19</lpage>. <pub-id pub-id-type="doi">10.1590/S0102-30982009000100002</pub-id></citation></ref>
<ref id="B2">
<label>2.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sorensen</surname> <given-names>HT</given-names></name> <name><surname>Sabroe</surname> <given-names>S</given-names></name> <name><surname>Olsen</surname> <given-names>J</given-names></name></person-group>. <article-title>A framework for evaluation of secondary data sources for epidemiological research</article-title>. <source>Int J Epidemiol.</source> (<year>1996</year>) <volume>25</volume>:<fpage>435</fpage>&#x02013;<lpage>42</lpage>.<pub-id pub-id-type="pmid">9119571</pub-id></citation></ref>
<ref id="B3">
<label>3.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Junior</surname> <given-names>AAG</given-names></name> <name><surname>Acurcio</surname> <given-names>FA</given-names></name> <name><surname>Reis</surname> <given-names>A</given-names></name> <name><surname>Santos</surname> <given-names>N</given-names></name> <name><surname>&#x000C1;vila</surname> <given-names>J</given-names></name> <name><surname>Dias</surname> <given-names>LV</given-names></name> <etal/></person-group>. <article-title>Building the National Database of Health Centred on the Individual: administrative and epidemiological record linkage - Brazil, 2000-2015</article-title>. <source>Int J Popul Data Sci.</source> (<year>2018</year>) <volume>3</volume>:<fpage>1</fpage>&#x02013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.23889/ijpds.v3i1.446</pub-id><pub-id pub-id-type="pmid">34095519</pub-id></citation></ref>
<ref id="B4">
<label>4.</label>
<citation citation-type="web"><source>Ci&#x000EA;ncia de Dados aplicada &#x000E0; Sa&#x000FA;de Plataforma de Ci&#x000EA;ncia de Dados aplicada &#x000E0; Sa&#x000FA;de</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://bigdata.icict.fiocruz.br/ciencia-de-dados-aplicada-saude">https://bigdata.icict.fiocruz.br/ciencia-de-dados-aplicada-saude</ext-link> (accessed December 10, 2020).</citation>
</ref>
<ref id="B5">
<label>5.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>Santar&#x000E9;m PRS</collab></person-group>. <source>Defini&#x000E7;&#x000E3;o de Dados Pessoais, Sens&#x000ED;veis e Anonimizados</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www2.camara.leg.br/atividade-legislativa/comissoes/comissoes-temporarias/especiais/55a-legislatura/pl-4060-12-tratamento-e-protecao-de-dados-pessoais/documentos/audiencias-e-eventos/paulo-rena-representante-do-instituto-beta-para-a-internet-e-democracia-ibidem">https://www2.camara.leg.br/atividade-legislativa/comissoes/comissoes-temporarias/especiais/55a-legislatura/pl-4060-12-tratamento-e-protecao-de-dados-pessoais/documentos/audiencias-e-eventos/paulo-rena-representante-do-instituto-beta-para-a-internet-e-democracia-ibidem</ext-link> (accessed December 10, 2020).</citation></ref>
<ref id="B6">
<label>6.</label>
<citation citation-type="web"><source>Rede Interagencial de Informa&#x000E7;&#x000F5;es para a Sa&#x000FA;de. Indicadores e Dados B&#x000E1;sicos - Brasil</source>. (<year>2012</year>). Available online at: <ext-link ext-link-type="uri" xlink:href="http://tabnet.datasus.gov.br/cgi/idb2012/matriz.htm">http://tabnet.datasus.gov.br/cgi/idb2012/matriz.htm</ext-link> (accessed December 10, 2020).</citation>
</ref>
<ref id="B7">
<label>7.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>Brasil</collab></person-group>. <article-title>Minist&#x000E9;rio da Sa&#x000FA;de. DATASUS</article-title>. <source>Servi&#x000E7;os</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="http://www2.datasus.gov.br/DATASUS/index.php?area=0901&#x00026;item=1">http://www2.datasus.gov.br/DATASUS/index.php?area=0901&#x00026;item=1</ext-link> (accessed December 10, 2020).</citation></ref>
<ref id="B8">
<label>8.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>Brasil</collab></person-group>. <article-title>Instituto Brasileiro de Geografia e Estat&#x000ED;stica - IBGE</article-title>. <source>Popula&#x000E7;&#x000E3;o</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.ibge.gov.br/estatisticas/sociais/populacao.html">https://www.ibge.gov.br/estatisticas/sociais/populacao.html</ext-link> (accessed December 10, 2020).</citation></ref>
<ref id="B9">
<label>9.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>Governo do Estado de S&#x000E3;o Paulo</collab></person-group>. <article-title>Governo Aberto SP</article-title>. <source>Conjunto de Dados para a Sociedade</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="http://www.governoaberto.sp.gov.br/">http://www.governoaberto.sp.gov.br/</ext-link> (accessed December 10, 2020).</citation></ref>
<ref id="B10">
<label>10.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>Brasil</collab></person-group>. <article-title>Governo Federal</article-title>. <source>Portal Brasileiro de Dados Abertos. Conjuntos de dados</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="http://dados.gov.br/dataset">http://dados.gov.br/dataset</ext-link> (accessed December 10, 2020).</citation></ref>
<ref id="B11">
<label>11.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Klein</surname> <given-names>RH</given-names></name> <name><surname>Klein</surname> <given-names>DCB</given-names></name> <name><surname>Luciano</surname> <given-names>EM</given-names></name></person-group>. <article-title>Identifica&#x000E7;&#x000E3;o de mecanismos para a amplia&#x000E7;&#x000E3;o da transpar&#x000EA;ncia em portais de dados abertos: uma an&#x000E1;lise no contexto brasileiro</article-title>. <source>Cad EBAPEBR.</source> (<year>2018</year>) <volume>16</volume>:<fpage>692</fpage>&#x02013;<lpage>715</lpage>. <pub-id pub-id-type="doi">10.1590/1679-395173241</pub-id></citation></ref>
<ref id="B12">
<label>12.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>CDC - NCHS - National Center for Health Statistics (2021)</collab></person-group>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.cdc.gov/nchs/index.htm">https://www.cdc.gov/nchs/index.htm</ext-link> (accessed April 28, 2021).</citation></ref>
<ref id="B13">
<label>13.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>CDC WONDER</collab></person-group>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://wonder.cdc.gov/WelcomeT.html">https://wonder.cdc.gov/WelcomeT.html</ext-link> (accessed April 28, 2021).</citation></ref>
<ref id="B14">
<label>14.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>RDC - Research Data Center Homepage</collab></person-group> (<year>2020</year>). Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.cdc.gov/rdc/index.htm1">https://www.cdc.gov/rdc/index.htm1</ext-link> (accessed April 28, 2021).</citation></ref>
<ref id="B15">
<label>15.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>Global Health Data Exchange | GHDx</collab></person-group>. Available online at: <ext-link ext-link-type="uri" xlink:href="http://ghdx.healthdata.org/">http://ghdx.healthdata.org/</ext-link> (accessed April 29, 2021).</citation></ref>
<ref id="B16">
<label>16.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>Brasil</collab></person-group>. <article-title>Instituto Brasileiro de Geografia e Estat&#x000ED;stica - IBGE</article-title>. <source>Sistemas de Estat</source>&#x000ED;<italic>sticas Vitais no Brasil: avan&#x000E7;os, perspectivas e desafios</italic>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.ibge.gov.br/estatisticas/sociais/populacao/21090-sistemas-de-estatisticas-vitais-no-brasil-avancos-perspectivas-e-desafios.html?=&#x00026;t=sobre">https://www.ibge.gov.br/estatisticas/sociais/populacao/21090-sistemas-de-estatisticas-vitais-no-brasil-avancos-perspectivas-e-desafios.html?=&#x00026;t=sobre</ext-link> (accessed December 10, 2020).</citation></ref>
<ref id="B17">
<label>17.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>Brasil</collab></person-group>. <article-title>Instituto Brasileiro de Geografia e Estat&#x000ED;stica - IBGE</article-title>. <source>Indicadores Sociais M</source>&#x000ED;<italic>nimos &#x02013; ISM</italic>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.ibge.gov.br/estatisticas/sociais/populacao/17374-indicadores-sociais-minimos.html?=&#x00026;t=resultados">https://www.ibge.gov.br/estatisticas/sociais/populacao/17374-indicadores-sociais-minimos.html?=&#x00026;t=resultados</ext-link> (accessed December 10, 2020).</citation></ref>
<ref id="B18">
<label>18.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>Rede Interagencial de Informa&#x000E7;&#x000E3;o para a Sa&#x000FA;de - RIPSA</collab></person-group>. <source>Indicadores b&#x000E1;sicos para a sa&#x000FA;de no Brasil: conceitos e aplica&#x000E7;&#x000F5;es</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="http://tabnet.datasus.gov.br/tabdata/livroidb/2ed/indicadores.pdf">http://tabnet.datasus.gov.br/tabdata/livroidb/2ed/indicadores.pdf</ext-link> (accessed December 10, 2020).</citation></ref>
<ref id="B19">
<label>19.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Duarte</surname> <given-names>CMR</given-names></name></person-group>. <article-title>Reflexos das pol&#x000ED;ticas de sa&#x000FA;de sobre as tend&#x000EA;ncias da mortalidade infantil no Brasil: revis&#x000E3;o da literatura sobre a &#x000FA;ltima d&#x000E9;cada</article-title>. <source>Cad Saude Publica.</source> (<year>2007</year>) <volume>23</volume>:<fpage>1511</fpage>&#x02013;<lpage>28</lpage>. <pub-id pub-id-type="doi">10.1590/S0102-311X2007000700002</pub-id> </citation></ref>
<ref id="B20">
<label>20.</label>
<citation citation-type="web"><source>Data Access - Vital Statistics Online</source>. (<year>2021</year>). Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.cdc.gov/nchs/data_access/vitalstatsonline.htm">https://www.cdc.gov/nchs/data_access/vitalstatsonline.htm</ext-link> (accessed April 28, 2021).</citation>
</ref>
<ref id="B21">
<label>21.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>Pan American Health Organization/World Health Organization (OPAS/OMS)</collab></person-group>. <source>Indicadores de Sa&#x000FA;de: Elementos Conceituais e Pr&#x000E1;ticos (Cap&#x000ED;tulo 2).</source> (<year>2018</year>). Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.paho.org/hq/index.php?option=com_content&#x00026;view=article&#x00026;id=14402:health-indicators-conceptual-and-operational-considerations-section-2&#x00026;Itemid=0&#x00026;showall=1&#x00026;lang=pt">https://www.paho.org/hq/index.php?option=com_content&#x00026;view=article&#x00026;id=14402:health-indicators-conceptual-and-operational-considerations-section-2&#x00026;Itemid=0&#x00026;showall=1&#x00026;lang=pt</ext-link> (accessed December 10, 2020).</citation></ref>
<ref id="B22">
<label>22.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Areco</surname> <given-names>KCN</given-names></name> <name><surname>Konstantyner</surname> <given-names>T</given-names></name> <name><surname>Taddei</surname> <given-names>JAAC</given-names></name></person-group>. <article-title>Tend&#x000EA;ncia secular da mortalidade infantil, componentes et&#x000E1;rios e evitabilidade no Estado de S&#x000E3;o Paulo &#x02013; 1996 a 2012</article-title>. <source>Rev Paul Pediatr.</source> (<year>2016</year>) <volume>34</volume>:<fpage>263</fpage>&#x02013;<lpage>70</lpage>. <pub-id pub-id-type="doi">10.1016/j.rpped.2016.01.006</pub-id></citation></ref>
<ref id="B23">
<label>23.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Victora</surname> <given-names>CG</given-names></name> <name><surname>Barros</surname> <given-names>FC</given-names></name></person-group>. <article-title>Infant mortality due to perinatal causes in Brazil: trends, regional patterns and possible interventions</article-title>. <source>S&#x000E3;o Paulo Med J.</source> (<year>2001</year>) <volume>119</volume>:<fpage>33</fpage>&#x02013;<lpage>42</lpage>. <pub-id pub-id-type="doi">10.1590/s1516-31802001000100009</pub-id><pub-id pub-id-type="pmid">11175624</pub-id></citation></ref>
<ref id="B24">
<label>24.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Coeli</surname> <given-names>CM</given-names></name></person-group>. <article-title>Sistemas de Informa&#x000E7;&#x000E3;o em Sa&#x000FA;de e uso de dados secund&#x000E1;rios na pesquisa e avalia&#x000E7;&#x000E3;o em sa&#x000FA;de</article-title>. <source>Cad Saude Colet.</source> (<year>2010</year>) <volume>18</volume>:<fpage>335</fpage>&#x02013;<lpage>6</lpage>.</citation></ref>
<ref id="B25">
<label>25.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Waldvogel</surname> <given-names>BC</given-names></name></person-group>. <article-title>Base unificada de nascimentos e &#x000F3;bitos no Estado de S&#x000E3;o Paulo: instrumento para aprimorar os indicadores de sa&#x000FA;de</article-title>. <source>S&#x000E3;o Paulo Perspect.</source> (<year>2008</year>) <volume>22</volume>:<fpage>161</fpage>.</citation></ref>
<ref id="B26">
<label>26.</label>
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Waldvogel</surname> <given-names>BC</given-names></name> <name><surname>Morais</surname> <given-names>LCC</given-names></name> <name><surname>Perdig&#x000E3;o</surname> <given-names>ML</given-names></name> <name><surname>Teixeira</surname> <given-names>MP</given-names></name> <name><surname>Freitas</surname> <given-names>RMV</given-names></name> <name><surname>Aranha</surname> <given-names>VJ</given-names></name></person-group>. <source>Experi&#x000EA;ncia da Funda&#x000E7;&#x000E3;o Seade com a aplica&#x000E7;&#x000E3;o da metodologia de vincula&#x000E7;&#x000E3;o determin&#x000ED;stica de bases de dados</source>. Ensaio &#x00026; Conjuntura. (<year>2019</year>) Available online at: <ext-link ext-link-type="uri" xlink:href="http://www.seade.gov.br/produtos/midia/2019/04/Ensaio_conjuntura_Vinculacao.pdf">http://www.seade.gov.br/produtos/midia/2019/04/Ensaio_conjuntura_Vinculacao.pdf</ext-link> (accessed December 10, 2020).</citation></ref>
<ref id="B27">
<label>27.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>Funda&#x000E7;&#x000E3;o Sistema Estadual de An&#x000E1;lise de Dados - Funda&#x000E7;&#x000E3;o SEADE</collab></person-group>. <source>Portal de Estat&#x000ED; do Estado de S&#x000E3;o Paulo. Sistema de Tabula&#x000E7;&#x000E3;o dos Microdados do Registro Civil para o Estado de S&#x000E3;o Paulo</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="http://produtos.seade.gov.br/produtos/mrc">http://produtos.seade.gov.br/produtos/mrc</ext-link> (accessed December 10, 2020).</citation></ref>
<ref id="B28">
<label>28.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>Brasil</collab></person-group>. <article-title>Lei Geral de Prote&#x000E7;&#x000E3;o de Dados (LGPD)</article-title>. <source>Lei n. 13.709, de 14 de agosto de 2018. Disp&#x000F5;e sobre a prote&#x000E7;&#x000E3;o de dados pessoais e altera a Lei n. 12.965 de 23 de abril de 2014 (Marco Civil da Internet). Di&#x000E1;rio Oficial da Uni&#x000E3;o, Bras&#x000ED;lia, 27 jul. 2020</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="http://www.planalto.gov.br/ccivil_03/_ato2015-2018/2018/lei/L13709.htm">http://www.planalto.gov.br/ccivil_03/_ato2015-2018/2018/lei/L13709.htm</ext-link> (accessed December 10, 2020).</citation></ref>
<ref id="B29">
<label>29.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Laureano</surname> <given-names>RMS</given-names></name> <name><surname>Caetano</surname> <given-names>N</given-names></name> <name><surname>Cortez</surname> <given-names>P</given-names></name></person-group>. <article-title>Previs&#x000E3;o de tempos de internamento num hospital portugu&#x000EA;s: aplica&#x000E7;&#x000E3;o da metodologia CRISP-DM</article-title>. <source>RISTI.</source> (<year>2014</year>) <volume>13</volume>:<fpage>83</fpage>&#x02013;<lpage>98</lpage>. <pub-id pub-id-type="doi">10.4304/risti.13.83-98</pub-id></citation></ref>
<ref id="B30">
<label>30.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chapman</surname> <given-names>P</given-names></name> <name><surname>Clinton</surname> <given-names>J</given-names></name> <name><surname>Kerber</surname> <given-names>R</given-names></name> <name><surname>Khabaza</surname> <given-names>T</given-names></name> <name><surname>Reinartz</surname> <given-names>T</given-names></name> <name><surname>Shearer</surname> <given-names>C</given-names></name> <etal/></person-group>. <source>CRIPS-DM 1.0 Step by Step Data Mining Guide</source>. <publisher-name>CRISP-DM Consortium</publisher-name> (<year>2000</year>).</citation></ref>
<ref id="B31">
<label>31.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Coeli</surname> <given-names>CM</given-names></name> <name><surname>Pinheiro</surname> <given-names>RS</given-names></name> <name><surname>Carvalho</surname> <given-names>MS</given-names></name></person-group>. <article-title>Nem melhor nem pior, apenas diferente</article-title>. <source>Cad Saude Publica.</source> (<year>2014</year>) <volume>30</volume>:<fpage>1363</fpage>&#x02013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1590/0102-311X00014814</pub-id> </citation></ref>
<ref id="B32">
<label>32.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Keinert</surname> <given-names>TMZ</given-names></name> <name><surname>Cortizo</surname> <given-names>CT</given-names></name></person-group>. <article-title>Dimens&#x000F5;es da privacidade das informa&#x000E7;&#x000F5;es em sa&#x000FA;de</article-title>. <source>Cad Saude Publica.</source> (<year>2018</year>) <volume>34</volume>:<fpage>e00039417</fpage>. <pub-id pub-id-type="doi">10.1590/0102-311X00039417</pub-id> </citation></ref>
<ref id="B33">
<label>33.</label>
<citation citation-type="web"><source>MySQL Community Edition</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.mysql.com/products/community/">https://www.mysql.com/products/community/</ext-link> (accessed December 10, 2020).</citation>
</ref>
<ref id="B34">
<label>34.</label>
<citation citation-type="web"><source>Microsoft&#x000AE; SQL Server&#x000AE; 2012 Express</source>. Microsoft Download Center. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.microsoft.com/pt-br/download/details.aspx?id=29062">https://www.microsoft.com/pt-br/download/details.aspx?id=29062</ext-link> (accessed December 10, 2020).</citation>
</ref>
<ref id="B35">
<label>35.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chiavegatto</surname> <given-names>Filho ADP</given-names></name></person-group>. <article-title>Uso de big data em sa&#x000FA;de no Brasil: perspectivas para um futuro pr&#x000F3;ximo</article-title>. <source>Epidemiol Serv Saude.</source> (<year>2015</year>) <volume>24</volume>:<fpage>325</fpage>&#x02013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.5123/S1679-49742015000200015</pub-id></citation></ref>
<ref id="B36">
<label>36.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>R Core Team</collab></person-group>. <source>R: A Language and Environment for Statistical Computing. R Foundation for Statistical Computing</source>. Vienna, Austria: R Core Team (<year>2012</year>) Available online at: <ext-link ext-link-type="uri" xlink:href="http://www.R-project.org/">http://www.R-project.org/</ext-link> (accessed December 10, 2020).</citation></ref>
<ref id="B37">
<label>37.</label>
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Vidmar</surname> <given-names>S</given-names></name> <name><surname>Stevens</surname> <given-names>L</given-names></name></person-group>. <source>Extracting Metadata from Stata Datasets</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.stata.com/meeting/oceania17/slides/oceania17_Vidmar.pdf">https://www.stata.com/meeting/oceania17/slides/oceania17_Vidmar.pdf</ext-link> (accessed December 10, 2020).</citation></ref>
<ref id="B38">
<label>38.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Galv&#x000E3;o</surname> <given-names>TF</given-names></name> <name><surname>Silva</surname> <given-names>MT</given-names></name> <name><surname>Garcia</surname> <given-names>LP</given-names></name></person-group>. <article-title>Ferramentas para melhorar a qualidade e a transpar&#x000EA;ncia dos relatos de pesquisa em sa&#x000FA;de: guias de reda&#x000E7;&#x000E3;o cient&#x000ED;fica</article-title>. <source>Epidemiol Serv Saude.</source> (<year>2016</year>) <volume>25</volume>:<fpage>427</fpage>&#x02013;<lpage>36</lpage>. <pub-id pub-id-type="doi">10.5123/S1679-49742016000200022</pub-id></citation></ref>
<ref id="B39">
<label>39.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ali</surname> <given-names>MS</given-names></name> <name><surname>Ichihara</surname> <given-names>MY</given-names></name> <name><surname>Lopes</surname> <given-names>LC</given-names></name> <name><surname>Barbosa</surname> <given-names>GCG</given-names></name> <name><surname>Pita</surname> <given-names>R</given-names></name> <name><surname>Carreiro</surname> <given-names>RP</given-names></name> <etal/></person-group>. <article-title>Administrative data linkage in Brazil: potentials for health technology assessment</article-title>. <source>Front Pharmacol.</source> (<year>2019</year>) <volume>10</volume>:<fpage>984</fpage>. <pub-id pub-id-type="doi">10.3389/fphar.2019.00984</pub-id><pub-id pub-id-type="pmid">31607900</pub-id></citation></ref>
<ref id="B40">
<label>40.</label>
<citation citation-type="web"><source>Funda&#x000E7;&#x000E3;o Oswaldo Cruz &#x0201C;Plataforma de Ci&#x000EA;ncia de Dados aplicada &#x000E0; Sa&#x000FA;de&#x0201D;</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://bigdata.icict.fiocruz.br/">https://bigdata.icict.fiocruz.br/</ext-link> (accessed December 10, 2020).</citation>
</ref>
<ref id="B41">
<label>41.</label>
<citation citation-type="web"><source>Microsoft Corporation &#x0201C;Microsoft Office 2010&#x0201D;</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.microsoft.com/en-us/microsoft-365/previous-versions/office-2010">https://www.microsoft.com/en-us/microsoft-365/previous-versions/office-2010</ext-link> (accessed December 10, 2020).</citation>
</ref>
<ref id="B42">
<label>42.</label>
<citation citation-type="book"><person-group person-group-type="author"><collab>World Health Organization</collab></person-group>. <source>International Statistical Classification of Diseases and Related Health Problems. 10th rev</source>. <publisher-loc>Geneve</publisher-loc>: <publisher-name>WHO</publisher-name> (<year>2010</year>).</citation></ref>
<ref id="B43">
<label>43.</label>
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Funda&#x000E7;&#x000E3;o</surname> <given-names>Sistema Estadual de An&#x000E1;lise de Dados - Funda&#x000E7;&#x000E3;o SEADE</given-names></name></person-group>. <source>Estrutura das bases de dados</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="http://www.seade.gov.br/editalfapesp/estruturasBasesNvObitosSeade.xlsx">http://www.seade.gov.br/editalfapesp/estruturasBasesNvObitosSeade.xlsx</ext-link> (accessed December 10, 2020).</citation></ref>
<ref id="B44">
<label>44.</label>
<citation citation-type="book"><person-group person-group-type="author"><collab>IBM Corp</collab></person-group>. <source>IBM SPSS Statistics for Windows, Version 24.0</source>. <publisher-loc>Armonk, NY</publisher-loc>: <publisher-name>IBM Corp</publisher-name> (<year>2016</year>).</citation></ref>
<ref id="B45">
<label>45.</label>
<citation citation-type="book"><person-group person-group-type="author"><collab>StataCorp</collab></person-group>. <source>Stata Statistical Software, Release 15</source>. <publisher-loc>College Station, TX</publisher-loc>: <publisher-name>StataCorp LLC</publisher-name> (<year>2017</year>).</citation></ref>
<ref id="B46">
<label>46.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Bott</surname> <given-names>E</given-names></name> <name><surname>Stinson</surname> <given-names>C</given-names></name></person-group>. <source>Windows 10 Inside Out</source>. <publisher-name>Microsoft Press</publisher-name> (<year>2019</year>).</citation></ref>
<ref id="B47">
<label>47.</label>
<citation citation-type="book"><person-group person-group-type="author"><collab>Circle Systems Inc</collab></person-group>. <source>Stat/Transfer, Version 6: File Transfer Utility for Windows.</source> <publisher-loc>Seattle</publisher-loc>: <publisher-name>Circle Systems</publisher-name> (<year>2000</year>).</citation></ref>
<ref id="B48">
<label>48.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>Brasil</collab></person-group>. <article-title>Instituto Brasileiro de Geografia e Estat&#x000ED;stica - IBGE</article-title>. <source>C&#x000F3;digos dos Munic</source>&#x000ED;<italic>pios</italic>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.ibge.gov.br/explica/codigos-dos-municipios.php">https://www.ibge.gov.br/explica/codigos-dos-municipios.php</ext-link> (accessed December 10, 2020).</citation></ref>
</ref-list>
<glossary>
<def-list>
<title>Abbreviations</title>
<def-item><term>CDC</term>
<def><p>Center for Disease Control and Prevention</p></def></def-item>
<def-item><term>CRISP-DM</term>
<def><p>Cross Industry Standard Process for Data Mining</p></def></def-item>
<def-item><term>DATASUS</term>
<def><p>Informatics Department of the Unified Health System</p></def></def-item>
<def-item><term>DC</term>
<def><p>Death Certificates</p></def></def-item>
<def-item><term>IBGE</term>
<def><p>Brazilian Institute of Geography and Statistics</p></def></def-item>
<def-item><term>LBC</term>
<def><p>Live Birth Certificates</p></def></def-item>
<def-item><term>PCDaS</term>
<def><p>Data Science Platform applied to Health</p></def></def-item>
<def-item><term>RDBMS</term>
<def><p>Relational Database Management Systems</p></def></def-item>
<def-item><term>RIPSA</term>
<def><p>Interagency Health Information Network</p></def></def-item>
<def-item><term>SEADE</term>
<def><p>S&#x000E3;o Paulo State Data Analysis System Foundation</p></def></def-item>
<def-item><term>SIM</term>
<def><p>Mortality Information System</p></def></def-item>
<def-item><term>SINASC</term>
<def><p>Live Birth Information System</p></def></def-item>
<def-item><term>SQL</term>
<def><p>Structured Query Language.</p></def></def-item>
</def-list>
</glossary>
<fn-group>
<fn fn-type="financial-disclosure"><p><bold>Funding.</bold> This study was funded by the Brazilian funding agency FAPESP: Projeto Tem&#x000E1;tico (Grant &#x00023; 2017/03748-7).</p>
</fn>
</fn-group>
</back>
</article> 