<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Phys.</journal-id>
<journal-title>Frontiers in Physics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Phys.</abbrev-journal-title>
<issn pub-type="epub">2296-424X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1387284</article-id>
<article-id pub-id-type="doi">10.3389/fphy.2024.1387284</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Physics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Symbol error rate minimization using deep learning approaches for short-reach optical communication networks</article-title>
<alt-title alt-title-type="left-running-head">Iqbal et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fphy.2024.1387284">10.3389/fphy.2024.1387284</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Iqbal</surname>
<given-names>Muhammad</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ghafoor</surname>
<given-names>Salman</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/970832/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ahmad</surname>
<given-names>Arsalan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2257560/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Aljohani</surname>
<given-names>Abdulah Jeza</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/971066/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Mirza</surname>
<given-names>Jawad</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2659397/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Aziz</surname>
<given-names>Imran</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Poti</surname>
<given-names>Luca</given-names>
</name>
<xref ref-type="aff" rid="aff7">
<sup>7</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Electrical Engineering and Computer Sciences (SEECS)</institution>, <institution>National University of Sciences and Technology</institution>, <addr-line>Islamabad</addr-line>, <country>Pakistan</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Electrical and Computer Engineering</institution>, <institution>King Abdulaziz University</institution>, <addr-line>Jeddah</addr-line>, <country>Saudi Arabia</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Electrical Engineering Department</institution>, <institution>HITEC University Taxila</institution>, <addr-line>Taxila</addr-line>, <country>Pakistan</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>SEECS Photonics Research Group</institution>, <addr-line>Islamabad</addr-line>, <country>Pakistan</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Department of Physics and Astronomy</institution>, <institution>Uppsala University</institution>, <addr-line>Uppsala</addr-line>, <country>Sweden</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Mirpur University of Science and Technology (MUST)</institution>, <addr-line>Mirpur</addr-line>, <country>Pakistan</country>
</aff>
<aff id="aff7">
<sup>7</sup>
<institution>National Inter-University Consortium for Telecommunications (CNIT)</institution>, Pisa, <country>Italy</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1387987/overview">Sushank Chaudhary</ext-link>, Chulalongkorn University, Thailand</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1420235/overview">Abhishek Sharma</ext-link>, Guru Nanak Dev University, India</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2063331/overview">Sunita Khichar</ext-link>, Chulalongkorn University, Thailand</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Imran Aziz, <email>imran.aziz@physics.uu.se</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>29</day>
<month>04</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>12</volume>
<elocation-id>1387284</elocation-id>
<history>
<date date-type="received">
<day>17</day>
<month>02</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>03</day>
<month>04</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Iqbal, Ghafoor, Ahmad, Aljohani, Mirza, Aziz and Poti.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Iqbal, Ghafoor, Ahmad, Aljohani, Mirza, Aziz and Poti</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Short-reach optical communication networks have various applications in areas where high-speed connectivity is needed, for example, inter- and intra-data center links, optical access networks, and indoor and in-building communication systems. Machine learning (ML) approaches provide a key solution for numerous challenging situations due to their robust decision-making, problem-solving, and pattern-recognition abilities. In this work, our focus is on utilizing deep learning models to minimize symbol error rates in short-reach optical communication setups. Various channel impairments, such as nonlinearity, chromatic dispersion (CD), and attenuation, are accurately modeled. Initially, we address the challenge of modeling a nonlinear channel. Consequently, we harness a deep learning model called autoencoders (AEs) to facilitate communication over nonlinear channels. Furthermore, we investigate how the inclusion of a nonlinear channel within an autoencoder influences the received constellation as the optical fiber length increases. Another facet of our work involves the deployment of a deep neural network-based receiver utilizing a channel influenced by chromatic dispersion. By gradually extending the optical length, we explore its impact on the received constellation and, consequently, the symbol error rate. Finally, we incorporate the split-step Fourier method (SSFM) to emulate the combined effects of nonlinearities, chromatic dispersion, and attenuation in the optical channel. This is accomplished through a neural network-based receiver. The outcome of this work is an evaluation and reduction of the symbol error rate as the length of the optical fiber is varied.</p>
</abstract>
<kwd-group>
<kwd>short-reach optical links</kwd>
<kwd>machine learning</kwd>
<kwd>optical access networks</kwd>
<kwd>symbol error rate</kwd>
<kwd>autoencoders</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Optics and Photonics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Over the last few years, there has been significant development in optical transmission systems to meet the increasing demands of the telecommunications sector. These advancements stem from the numerous advantages of optical fiber, which include swifter transmission, reduced signal loss (attenuation), smaller physical dimensions, heightened resistance to electromagnetic interference, and greater capacity for data transmission. Presently, there is substantial interest in short-distance communication systems within both industry and academia. The term &#x201c;short-reach&#x201d; pertains to communication setups with transmission distances under 100&#xa0;km, which are cost-sensitive due to their widespread deployment [<xref ref-type="bibr" rid="B1">1</xref>, <xref ref-type="bibr" rid="B2">2</xref>]. Short-reach optical communication encounters more intricate challenges owing to factors like the advent of 5G technology, the envisioned developments beyond 5G (referred to as 6G), the utilization of edge-distributed cloud computing networks, extensive machine-to-machine communications, communication within or between data centers, and mobile front-haul setups [<xref ref-type="bibr" rid="B1">1</xref>, <xref ref-type="bibr" rid="B3">3</xref>, <xref ref-type="bibr" rid="B4">4</xref>]. For the sake of discussion, short-reach optical networks can be classified into five categories based on their transmission distance and function: inter-data center networks, intra-data center networks, optical access networks, indoor and in-building optical wireless communications, and mobile front-haul communications. Communication requirements are becoming progressively rigorous, and the intricacy of short-reach optical networks grows significantly across these various types. To tackle these challenges, there is a growing proposal and extensive study of incorporating artificial intelligence (AI) in short-reach optical systems. AI emulates biological processes like learning, self-correction, and extrapolation, enabling computers to manage complex scenarios. In recent decades, the application of AI to enhance the performance of optical networks and systems has emerged as a prominent area of research, encompassing network management and transmission. Machine learning (ML), a subset of AI, empowers an agent to enhance its future task execution by learning from past experiences.</p>
<p>ML techniques have garnered significant attention within the realm of short-reach optical communication systems due to their aptitude for problem-solving, decision-making, and pattern recognition. Their application is widespread in various aspects of short-reach optical systems, encompassing tasks such as optical performance monitoring (OPM), signal processing, modulation format identification (MFI), and indoor optical wireless systems.</p>
<p>In the context of short-reach optical systems that use compact and cost-effective components, such as those utilizing direct detection receivers, achieving desired data rates in a cost-efficient manner requires innovative signaling and digital signal processing (DSP) techniques. As an alternative to conventional DSP methods, ML algorithms are emerging as effective solutions for addressing nonlinear challenges. Neural networks can also play a role in optimizing transmitter or receiver DSP functions [<xref ref-type="bibr" rid="B5">5</xref>]. In short-reach optical networks based on intensity modulation/direct detection (IM/DD) techniques, which are known for their cost-effectiveness, photodiodes are used to detect nonlinear signals. The presence of dispersion can lead to significant impairments, notably intersymbol interference (ISI). To combat these challenges and enhance system performance, ML techniques are gaining traction as viable alternatives to traditional DSP approaches [<xref ref-type="bibr" rid="B1">1</xref>, <xref ref-type="bibr" rid="B6">6</xref>]. These ML-based techniques, such as equalization, autoencoders (AEs), digital pre-distortion, and soft-demapping, have been employed in various ways, encompassing compensation for both linear and nonlinear distortions (NLDs).</p>
<p>To tackle the challenge of compensating for both linear and nonlinear distortions in optical networks, advanced techniques such as maximum likelihood sequence estimation (MLSE) and equalizers based on ML algorithms are under exploration [<xref ref-type="bibr" rid="B7">7</xref>]. Among these approaches, ML-based methods, particularly those leveraging neural networks, are gaining recognition for their effectiveness in equalization tasks. Neural networks are being employed either to assist other signal processing stages or directly as comprehensive equalizers. These studies highlight that neural networks outperform traditional equalizers, positioning them as a sought-after technology with significant value in short-reach applications [<xref ref-type="bibr" rid="B8">8</xref>]. Various architectural designs have been proposed for this purpose, including feedforward neural networks (FFNNs) [<xref ref-type="bibr" rid="B9">9</xref>&#x2013;<xref ref-type="bibr" rid="B11">11</xref>], reservoir computing (RC) [<xref ref-type="bibr" rid="B12">12</xref>&#x2013;<xref ref-type="bibr" rid="B18">18</xref>], and recurrent neural networks (RNNs) [<xref ref-type="bibr" rid="B19">19</xref>, <xref ref-type="bibr" rid="B20">20</xref>]. These architectures can be effectively utilized for both linear and nonlinear equalization tasks.</p>
<p>Similarly, there exist signal processing methods such as autoencoders [<xref ref-type="bibr" rid="B21">21</xref>&#x2013;<xref ref-type="bibr" rid="B24">24</xref>] that facilitate the development of end-to-end processes where the transmitter, channel, and receiver are jointly optimized. This idea was first introduced for wireless communication [<xref ref-type="bibr" rid="B25">25</xref>&#x2013;<xref ref-type="bibr" rid="B27">27</xref>] and later quickly utilized for optical fiber communications [<xref ref-type="bibr" rid="B6">6</xref>, <xref ref-type="bibr" rid="B21">21</xref>]. Autoencoders comprise three key components: the encoder, code, and decoder. The encoder reduces input dimensionality, and the decoder restores the reduced code dimension. This process retains essential features of the input data after dimensionality reduction, leading to reduced transmission rates and enhanced communication reliability [<xref ref-type="bibr" rid="B21">21</xref>]. A proposal involving fully connected neural networks and a bidirectional LSTM (BiLSTM) model for the channel introduces an autoencoder for intensity modulation/direct detection (IM/DD) systems [<xref ref-type="bibr" rid="B22">22</xref>]. This auto-encoder capitalizes on optical signal characteristics in the time domain and incorporates various system factors like nonlinearity, attenuation, dispersion, and optical-to-electrical square-law conversion. The primary goal of the auto-encoder is to minimize the dimensions of the input signal, thereby enhancing system reliability and reducing transmission rates while preserving essential features.</p>
<p>When addressing impairments such as linear and nonlinear distortions in power amplifiers, a widely used technique is digital pre-distortion (DPD), which commonly employs Volterra-based algorithms involving intricate direct and indirect learning architectures [<xref ref-type="bibr" rid="B28">28</xref>]. To offer a lower-complexity alternative to Volterra-based algorithms, an approach using an extreme learning machine (ELM) for DPD has been proposed. This method enables rapid estimation and compensation of transfer functions, including those of Mach&#x2013;Zehnder modulators (MZMs) [<xref ref-type="bibr" rid="B29">29</xref>]. In the realm of DPD, an ML-based approach utilizing FFNNs has been recommended to enhance the power efficiency of radio-over-fiber (RoF) links in analog optical front-haul applications [<xref ref-type="bibr" rid="B28">28</xref>].</p>
<p>For achieving high spectral efficiency and speed in optical communication networks, technologies like soft decision forward error correction (FEC) and higher-order quadrature amplitude modulation (QAM) have been studied. When dealing with nonlinear equalization (NLE) and soft decision de-mapping in Volterra-based equalization, complexity can be a challenge. An alternative approach employs a soft neural network (NN) based method known as soft deep neural network (SDNN) architecture. This approach is explored for its ability to address the temporal dynamic behavior of sequences of time, providing an effective solution to nonlinearities [<xref ref-type="bibr" rid="B30">30</xref>]. Likewise, a soft de-mapper based on bidirectional RNN techniques has been suggested [<xref ref-type="bibr" rid="B31">31</xref>] in the context of improving performance compared to a soft de-mapper based on artificial neural networks (ANNs) [<xref ref-type="bibr" rid="B30">30</xref>, <xref ref-type="bibr" rid="B32">32</xref>]. This approach capitalizes on the ability of bidirectional RNNs to represent infinite impulse responses and nonlinearities, thereby effectively capturing the temporal dynamic behavior of sequences of time.</p>
<p>This study focuses on employing deep learning models to reduce symbol error rates in short-distance optical communication configurations. We accurately simulate various channel impairments like nonlinearity, CD, and attenuation. Initially, we tackle the task of modeling a nonlinear channel. Subsequently, we use autoencoders, a type of deep learning model, to aid communication over nonlinear channels. Additionally, we examine how integrating a nonlinear channel into an autoencoder affects the received constellation as the length of the optical fiber increases. Another aspect of our research involves implementing a receiver based on deep neural networks, which operate in a channel influenced by chromatic dispersion. By gradually increasing the optical length, we analyze its influence on the received constellation and, consequently, the symbol error rate. Lastly, we introduce the split-step Fourier method (SSFM) to simulate the combined impacts of nonlinearities, chromatic dispersion, and attenuation in the optical channel. This simulation is achieved through a neural network-based receiver approach. The remainder of this paper is structured as follows: in <xref ref-type="sec" rid="s2">Section 2</xref>, we describe the basic concepts of deep learning and autoencoders. The proposed architecture of the autoencoder is presented in <xref ref-type="sec" rid="s3">Section 3</xref>. <xref ref-type="sec" rid="s4">Section 4</xref> explains the simulation scenarios based on different channel impairments along with their effect on model performance. In Section 5, the split-step Fourier method and its simulation parameters are discussed. Finally, the conclusion and future work directions are given in Section 6.</p>
</sec>
<sec id="s2">
<title>2 Background</title>
<p>This section introduces the DNNs [<xref ref-type="bibr" rid="B33">33</xref>, <xref ref-type="bibr" rid="B34">34</xref>] and AEs.</p>
<sec id="s2-1">
<title>2.1 Fully connected deep neural networks</title>
<p>A feedforward model, such as a <italic>N</italic>-layer fully connected DNN, maps an input vector <italic>v</italic>
<sub>0</sub> to an output vector <italic>v</italic>
<sub>
<italic>N</italic>
</sub> &#x003D; <italic>f</italic>
<sub>
<italic>DNN</italic>
</sub>(<italic>v</italic>
<sub>0</sub>) through repetitive steps of the form is given by<disp-formula id="e1">
<mml:math id="m1">
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x003D;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x002B;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="2em"/>
<mml:mi>n</mml:mi>
<mml:mo>&#x003D;</mml:mo>
<mml:mn>1,2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>.</mml:mo>
<mml:mi>N</mml:mi>
<mml:mo>,</mml:mo>
</mml:math>
<label>(1)</label>
</disp-formula>where <italic>v</italic>
<sub>
<italic>n</italic>&#x2212;1</sub> is the output of (<italic>n</italic> &#x2212; 1)th layer, <italic>v</italic>
<sub>
<italic>n</italic>
</sub> is the output of <italic>n</italic>th layer, <bold>W</bold>
<sub>
<italic>n</italic>
</sub> is the weight matrix, <bold>b</bold>
<sub>
<italic>n</italic>
</sub> is the bias vector of <italic>n</italic>th layer, and <italic>&#x3b1;</italic>
<sub>
<italic>n</italic>
</sub> is the activation layer [<xref ref-type="bibr" rid="B21">21</xref>]. The parameters <bold>W</bold>
<sub>
<italic>n</italic>
</sub> and <bold>b</bold>
<sub>
<italic>n</italic>
</sub> of the layer are represented by<disp-formula id="e2">
<mml:math id="m2">
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x003D;</mml:mo>
<mml:mfenced close="}" open="{">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
<p>The network is capable of approximating nonlinear functions through the use of the activation function <italic>&#x3b1;</italic>
<sub>
<italic>n</italic>
</sub>, which establishes nonlinear relationships between the layers. The rectified linear unit (ReLU) is a frequently used activation function in modern ANNs. It acts separately on each of its input vector elements by maintaining the positive values and equating negative to zero [<xref ref-type="bibr" rid="B35">35</xref>], i.e., <italic>z</italic> &#x003D; <italic>&#x3b1;</italic>
<sub>
<italic>Relu</italic>
</sub>(<italic>k</italic>) with<disp-formula id="e3">
<mml:math id="m3">
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x003D;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:math>
<label>(3)</label>
</disp-formula>where <italic>z</italic> and <italic>k</italic> are vectors whose <italic>i</italic>th elements are represented by <italic>z</italic>
<sub>
<italic>i</italic>
</sub> and <italic>k</italic>
<sub>
<italic>i</italic>
</sub>, respectively. The ReLU function has a continuous gradient when compared to other well-known activation functions like the hyperbolic tangent and sigmoid, which makes training computationally less expensive and prevents the effect of vanishing gradients. This impact happens for activation functions with asymptotic behavior because the gradient can shrink and slow the learning algorithm&#x2019;s convergence.</p>
<p>The softmax activation function is frequently used in the last (decision) layer of an ANN, and the elements <italic>z</italic>
<sub>
<italic>i</italic>
</sub> of the output <italic>z</italic> &#x003D; softmax <italic>k</italic>) are provided by<disp-formula id="e4">
<mml:math id="m4">
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x003D;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>p</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>p</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:math>
<label>(4)</label>
</disp-formula>
</p>
<p>By labeling the training data, the neural network training may be performed under supervision. This establishes a pairing between the intended output vector <inline-formula id="inf1">
<mml:math id="m5">
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> and input vector <italic>v</italic>
<sub>0</sub>. As a result, the training goal is to minimize the cost function <inline-formula id="inf2">
<mml:math id="m6">
<mml:mi mathvariant="script">C</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> as a function of the weight matrix, <bold>W</bold>
<sub>
<italic>n</italic>
</sub>, and bias vector, <bold>b</bold>
<sub>
<italic>n</italic>
</sub>, of all <italic>N</italic> layers. The cost function over the set of training inputs <italic>M</italic> between the DNN output, <italic>v</italic>
<sub>
<italic>N</italic>
</sub>, and intended output,<inline-formula id="inf3">
<mml:math id="m7">
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>, is given by<disp-formula id="e5">
<mml:math id="m8">
<mml:mi mathvariant="script">C</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x003D;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>M</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x003D;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mi>L</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<p>In Eq. <xref ref-type="disp-formula" rid="e5">5</xref>, <italic>L</italic> (<italic>z</italic>, <italic>k</italic>) represents the loss function and &#x7c;<italic>M</italic>&#x7c; is the training example number. Cross-entropy is the loss function that we used in this work, represented as<disp-formula id="e6">
<mml:math id="m9">
<mml:mi>L</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x003D;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>.</mml:mo>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>In modern deep learning, error backpropagation allows for efficient computation of the gradient [<xref ref-type="bibr" rid="B33">33</xref>]. Adam optimizer, a cutting-edge algorithm with improved convergence, dynamically changes the learning rate <italic>&#x3b7;</italic> [<xref ref-type="bibr" rid="B36">36</xref>]. In this work, training process optimization is carried out using the Adam algorithm, and PyTorch and Keras [<xref ref-type="bibr" rid="B37">37</xref>] were used to create all of the numerical results.</p>
</sec>
<sec id="s2-2">
<title>2.2 Basics of autoencoder</title>
<p>An AE is a concept in which data are first encoded into a compressed form and subsequently decoded to revert it to its original shape. This technique, employed in unsupervised learning, involves acquiring a condensed representation of input data through the utilization of an NN. This acquired representation can then be effectively applied to tasks such as reducing noise, compressing data, or extracting features. The autoencoder is composed of three primary components: the encoder section, a code, and the decoder section. The encoder element takes the input data and compresses it into a lower-dimensional form referred to as the encoded data or code. This compressed representation is usually presented as a vector of numerical values, serving as a compact summary of the input data. This implies that the encoder&#x2019;s function is to compress the data. Within the encoder, one or more hidden layers exist that apply nonlinear transformations to the data, generating the encoded version. The structure of an autoencoder is depicted in <xref ref-type="fig" rid="F1">Figure 1</xref>. Subsequently, the code is input into the decoder phase, which effectively reconstructs the initial data, namely, the input data, from the dimensionally reduced code. This decoder is comprised of one or more concealed layers that undertake the task of transforming the condensed representation back into the original data form. The objective of the decoder&#x2019;s output is to closely mirror the input data. The primary aim is to restore the original data with minimal loss of information from the compressed representation.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Structure of an autoencoder.</p>
</caption>
<graphic xlink:href="fphy-12-1387284-g001.tif"/>
</fig>
<p>A loss function is employed to minimize disparities between the input and reconstructed data, often using metrics like mean squared error, during the training process of the AE. The weights of both the encoder and decoder are updated through the utilization of backpropagation and optimization algorithms like stochastic gradient descent. <xref ref-type="fig" rid="F2">Figure 2</xref> illustrates the sequential progression of the autoencoder. Beginning with the input data, the autoencoder undertakes the task of learning how to extract the most pivotal features and subsequently encapsulate them in a streamlined manner. This is achieved by implementing a bottleneck layer within the network&#x2019;s core, housing fewer neurons in comparison to the input and output layers. Through this method of data compression, the autoencoder gains the ability to understand the process of discarding irrelevant information while concentrating on the most significant and critical features. Upon the successful training of the autoencoder, the code generated by the encoder can be harnessed for various purposes, such as data compression or feature extraction. For instance, if the autoencoder has been trained on images, the compact representation of the image, i.e., the code produced by the encoder, can be applied for tasks like image retrieval or classification.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Flow process of an AE.</p>
</caption>
<graphic xlink:href="fphy-12-1387284-g002.tif"/>
</fig>
<p>Autoencoders possess the remarkable capacity to acquire valuable data representations devoid of the necessity for labeled data, which stands out as a significant advantage. This attribute renders them particularly advantageous for tasks where acquiring labeled data is resource-intensive or challenging. Moreover, autoencoders exhibit proficiency in handling high-dimensional data, a domain often fraught with challenges for other categories of machine learning algorithms. Nonetheless, grappling with overfitting remains one of the principal hurdles associated with autoencoders. This occurs when the autoencoder predominantly learns the intricacies of the training data features rather than cultivating valuable, generalizable features suitable for new data instances. To counteract the menace of overfitting, regularization approaches such as dropout techniques and weight decay methods can be harnessed. These strategies serve to maintain the autoencoder&#x2019;s focus on salient and broadly applicable features, curbing the inclination toward excessive adaptation to training data specifics.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Proposed end-to-end communication system</title>
<p>We put into practice a fiber optic communication system and transmission chain, including the transmitter, receiver, and channel, as a full end-to-end ANN, as suggested in [<xref ref-type="bibr" rid="B26">26</xref>, <xref ref-type="bibr" rid="B27">27</xref>]. The above-stated concept has been extended to the interpretation of communication system components, consisting of the transmitting part, channel, and receiving part. <xref ref-type="fig" rid="F3">Figure 3</xref> shows the basic components of the communication system, which consists of the transmitter, channel, and receiver.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Illustration of the proposed autoencoder.</p>
</caption>
<graphic xlink:href="fphy-12-1387284-g003.tif"/>
</fig>
<p>The proposed autoencoder structure is shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. A message <italic>s</italic> is sent, which is chosen from a pool of <italic>M</italic> possible messages {1, 2, &#x2026; , <italic>M</italic>}&#x225c;<bold>
<italic>M</italic>
</bold>. For every message, <italic>log</italic>
<sub>2</sub>M bits are represented. In accordance with (30), first the messages are converted into &#x201c;one hot encoded&#x201d; vectors of dimension <italic>M</italic>, where 1 is the <italic>sth</italic> element and the other remaining elements are represented by 0. These vectors are fed as an input to transmitter NN, which contains numerous layers of densely connected neurons. Every neuron receives inputs from the layer before it and produces an output based on <italic>z</italic>
<sub>
<italic>out</italic>
</sub> &#x003D; <italic>f</italic> (<bold>w</bold>
<sup>
<bold>T</bold>
</sup>
<bold>z</bold>
<sub>
<bold>in</bold>
</sub> &#x002B; <italic>b</italic>), where <bold>w</bold> represents a vector of weights, b &#x2208; R denotes a bias, and <italic>f</italic> (&#x22c5;) is the activation function, which is assumed to be linear in this case.</p>
<p>The transmitter&#x2019;s two outputs <italic>z</italic>
<sub>
<italic>i</italic>
</sub> and <italic>z</italic>
<sub>
<italic>r</italic>
</sub> are used as inputs to the channel. To satisfy the constraint of average power, the NN is normalized by the use of <italic>M</italic> different training inputs. The input of the channel, denoted as <italic>x</italic>, is selected randomly by a constellation consisting of M points with a second moment <italic>E</italic> [&#x7c;<italic>x</italic>&#x7c;<sup>2</sup>] &#x003D; <italic>P</italic>
<sub>
<italic>in</italic>
</sub>, where <italic>P</italic>
<sub>
<italic>in</italic>
</sub> is equivalent to the input power. After that, the normalized output is transmitted through the channel, and <italic>y</italic> output is produced. The channel output <italic>y</italic> consisting of real <italic>y</italic>
<sub>
<italic>r</italic>
</sub> element and imaginary <italic>y</italic>
<sub>
<italic>i</italic>
</sub> element are fed to a receiver NN as an input. The receiver NN outputs a probability distribution <italic>f</italic>
<sub>
<italic>y</italic>
</sub> (<italic>s</italic>&#x2032;) in [0,1], <italic>s</italic>&#x2032; &#x2208; <bold>
<italic>M</italic>
</bold> over all possible transmitted messages represented by set <bold>
<italic>M</italic>
</bold>. The output is normalized to ensure that the sum of all probabilities should be equal to 1. The estimated transmitted message <inline-formula id="inf4">
<mml:math id="m10">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> is then obtained by selecting the message with the highest probability from the output distribution, given by <italic>s</italic> &#x003D; <italic>argmax</italic>
<sub>
<italic>s</italic>&#x2032;</sub>, <italic>f</italic>
<sub>
<italic>y</italic>
</sub> (<italic>s</italic>&#x2032;) [<xref ref-type="bibr" rid="B38">38</xref>].</p>
</sec>
<sec id="s4">
<title>4 Simulation scenarios and results</title>
<p>In this section, different channel impairments have been modeled and their effects on the communication system have been explained.</p>
<sec id="s4-1">
<title>4.1 Nonlinearity based channel</title>
<p>First, we consider the effects of nonlinear phase noise in single-mode optic fiber models. The nonlinear Schr&#xf6;dinger equation (NLSE) is used for modeling the propagation of signals in an optical fiber employing distributed amplification as follows [<xref ref-type="bibr" rid="B39">39</xref>]:<disp-formula id="e7">
<mml:math id="m11">
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>k</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x003D;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mi>x</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:math>
<label>(7)</label>
</disp-formula>
</p>
<p>Here, <italic>k</italic> (<italic>d</italic>, <italic>t</italic>) is the signal, which is transmitted, <italic>t</italic> is the time coordinate, and <italic>d</italic> is the distance coordinate; the nonlinearity parameter is represented by <italic>&#x3b3;</italic>, <italic>&#x3b2;</italic>
<sub>2</sub> is the group velocity dispersion (GVD) coefficient, and the Gaussian noise is represented by <italic>p</italic> (<italic>d</italic>, <italic>t</italic>). The right side of the equation has two terms. The equation&#x2019;s first term shows the Kerr effect, which induces a shift in the phase that is proportional to the signal&#x2019;s power and leads to a substantial distortion in optical fiber systems. The second term shows dispersion. As the channel likelihood function in [Eq. <xref ref-type="disp-formula" rid="e7">7</xref>] is unknown, a simple dispersionless channel is considered by ignoring <italic>&#x3b2;</italic>
<sub>2</sub> in [Eq. <xref ref-type="disp-formula" rid="e7">7</xref>]. The equation of the model based on recursion is given as follows [<xref ref-type="bibr" rid="B40">40</xref>]:<disp-formula id="e8">
<mml:math id="m12">
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x002B;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x003D;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>k</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>/</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x002B;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x002B;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mspace width="2em"/>
<mml:mn>0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>.</mml:mo>
</mml:math>
<label>(8)</label>
</disp-formula>
</p>
<p>Here, <italic>k</italic>
<sub>0</sub> &#x003D; <italic>k</italic> is the input to the channel consisting of complex values, <italic>r</italic> &#x003D; <italic>k</italic>
<sub>
<italic>S</italic>
</sub> is the output of the channel, <italic>p</italic>
<sub>
<italic>i</italic>&#x002B;1</sub> &#x223c; <italic>CN</italic>(0, <italic>P</italic>
<sub>
<italic>N</italic>
</sub>/<italic>S</italic>), fiber link length is represented by <italic>L</italic>
<sub>
<italic>o</italic>
</sub>, <italic>&#x3b3;</italic> is the nonlinearity parameter, and <italic>P</italic>
<sub>
<italic>N</italic>
</sub> is the noise power. Ideal distributed amplification is assumed by the model and <italic>S</italic> &#x2192; <italic>&#x221e;</italic>.</p>
</sec>
<sec id="s4-2">
<title>4.2 Learned constellations</title>
<p>To obtain numerical results, the fiber model in [Eq. <xref ref-type="disp-formula" rid="e8">8</xref>] is assumed to have an optical fiber length of <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 100&#xa0;km, a nonlinearity coefficient of <italic>&#x3b3;</italic> &#x003D; 1.27, and a noise power of <italic>P</italic>
<sub>
<italic>N</italic>
</sub> &#x003D; &#x2212;21.3 dBm. The model is iterated S &#x003D; 50 times for simulation, providing a good approximation of the true asymptotic channel PDF [<xref ref-type="bibr" rid="B41">41</xref>].</p>
<p>Using PyTorch&#x2019;s random number generator, the dataset gets generated inside the network on the fly. The training dataset size is 1.10<sup>4</sup>. The number of Epochs &#x003D; 120, and the batch size is varied during the training. The size of the validation dataset is 1.10<sup>5</sup>. The cross-entropy loss function and Adam optimizer [<xref ref-type="bibr" rid="B36">36</xref>] in PyTorch are used for training the AE separately for various values of <italic>P</italic>
<sub>
<italic>in</italic>
</sub>. The AE structure parameters for M &#x003D; 16 are summarized in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Autoencoder parameters for M &#x003D; 16.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center" colspan="3">Transmitter</th>
<th align="center" colspan="3">Receiver</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Layers</td>
<td align="center">1</td>
<td align="center">2</td>
<td align="center">1</td>
<td align="center">2&#x2013;3</td>
<td align="center">4</td>
</tr>
<tr>
<td align="center">Neurons</td>
<td align="center">16</td>
<td align="center">2</td>
<td align="center">2</td>
<td align="center">50</td>
<td align="center">16</td>
</tr>
<tr>
<td align="center">Activation function</td>
<td align="center">-</td>
<td align="center">Identity</td>
<td align="center">ELU</td>
<td align="center">ELU</td>
<td align="center">ELU</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In <xref ref-type="fig" rid="F4">Figure 4</xref>, the 16-point constellation that has been learned under different <italic>P</italic>
<sub>
<italic>in</italic>
</sub> for nonlinearity-based fiber optic channels is shown. The points in the learned constellation are equally spaced and pentagonal-shaped at &#x2212;11 dBm, as shown in <xref ref-type="fig" rid="F4">Figure 4A</xref>. In highly nonlinear regimes, the constellations appear to be random. The constellation points with high energy have varying radii. As a result, after propagation through the nonlinear fiber channel, these points do not overlap with other points. Additionally, the farther the point is from others, the larger its radius.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Nonlinear fiber channel 16-point learned constellations for different <italic>P</italic>
<sub>
<italic>in</italic>
</sub> values. <bold>(A)</bold> <italic>P</italic>
<sub>
<italic>in</italic>
</sub> &#x003D; &#x2212;11 dBm. <bold>(B)</bold> <italic>P</italic>
<sub>
<italic>in</italic>
</sub> &#x003D; &#x2212;1 dBm. <bold>(C)</bold> <italic>P</italic>
<sub>
<italic>in</italic>
</sub> &#x003D; 6 dBm. <bold>(D)</bold> <italic>P</italic>
<sub>
<italic>in</italic>
</sub> &#x003D; 12 dBm.</p>
</caption>
<graphic xlink:href="fphy-12-1387284-g004.tif"/>
</fig>
</sec>
<sec id="s4-3">
<title>4.3 Effect of the nonlinear channel on fiber length</title>
<p>In a nonlinear channel, when the length of the optical fiber increases, the nonlinearity of the fiber increases as well, leading to signal quality degradation, which can limit the achievable transmission distance and data rate. For obtaining different numerical results, consider the optic fiber model in [Eq. <xref ref-type="disp-formula" rid="e8">8</xref>] with a nonlinearity coefficient of <italic>&#x3b3;</italic> &#x003D; 1.27, noise power of <italic>P</italic>
<sub>
<italic>N</italic>
</sub> &#x003D; &#x2212;21.3 dBm, and input power of <italic>P</italic>
<sub>
<italic>in</italic>
</sub> &#x003D; 2 dBm. <xref ref-type="table" rid="T1">Table 1</xref> shows the autoencoder parameters used for different lengths. For training the model, the number of Epochs &#x003D; 120, and the batch size is varied during the training. In the initial iteration, the batch size is kept small to obtain a working solution. The size of the batch increases as the training nears completion. If the size of the batch is kept small, there will be no misclassifications, and the training will not improve. If the batch size is large, there are higher chances of error in the batch; hence, there will be a reason for the training to keep on improving. The results of training loss curves are shown for different fiber lengths in <xref ref-type="fig" rid="F5">Figure 5</xref>. The loss function called cross-entropy and the optimizer known as Adam [<xref ref-type="bibr" rid="B36">36</xref>] are used during the training in PyTorch.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Training loss curves for different optical fiber lengths.</p>
</caption>
<graphic xlink:href="fphy-12-1387284-g005.tif"/>
</fig>
<p>The effect of nonlinearity for different optical fiber lengths, <italic>L</italic>
<sub>
<italic>o</italic>
</sub>, is shown in <xref ref-type="fig" rid="F6">Figure 6</xref>. The effect of the nonlinear channel on <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 5,000&#xa0;km is shown in <xref ref-type="fig" rid="F6">Figure 6A</xref>. The received constellation is affected by phase noise or phase rotation. The right side of the figure shows the decision regions formed on the basis of the received constellation. When the length is decreased to <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 100&#xa0;km, the nonlinear channel has less impact on the received symbols as compared to <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 5,000&#xa0;km. In other words, the received constellation has less phase noise, and decision regions can easily segregate the received symbols, as shown in <xref ref-type="fig" rid="F6">Figure 6B</xref>. Length is reduced further to <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 50&#xa0;km and <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 15&#xa0;km, as shown in <xref ref-type="fig" rid="F6">Figure 6C</xref> and <xref ref-type="fig" rid="F6">6D</xref>. The received signal is less affected by the phase noise, and more accurate information is obtained at the receiver side with a reduced symbol error rate. This effect can also be seen in the decision boundaries formed on the received constellation. Therefore, when the fiber length is increased, nonlinearity will affect the received constellation by increasing the phase noise. In other words, received information will be degraded, and the symbol error rate will increase, making it difficult for the receiver to recover the signal information. The symbol error rate on the validation dataset for different optical fibers is shown in <xref ref-type="fig" rid="F7">Figure 7</xref>.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Received constellations and decision regions for different optical fiber lengths. <bold>(A)</bold> <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 5,000&#xa0;km. <bold>(B)</bold> <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 100&#xa0;km. <bold>(C)</bold> <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 50&#xa0;km. <bold>(D)</bold> <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 15&#xa0;km.</p>
</caption>
<graphic xlink:href="fphy-12-1387284-g006.tif"/>
</fig>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>SER on the validation dataset for different optical fiber lengths. <bold>(A)</bold> <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 5,000&#xa0;km. <bold>(B)</bold> <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 100&#xa0;km. <bold>(C)</bold> <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 50&#xa0;km. <bold>(D)</bold> <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 15&#xa0;km.</p>
</caption>
<graphic xlink:href="fphy-12-1387284-g007.tif"/>
</fig>
</sec>
<sec id="s4-4">
<title>4.4 Chromatic dispersion based channel</title>
<p>We consider the effects of chromatic dispersion-based channels on the fiber length. First, a discrete Fourier transform (DFT) is used to convert the original signal into the frequency domain. 16-QAM is used as input. It is fed into the channel, and chromatic dispersion is applied to it. To shift to the time domain (TD), the inverse discrete Fourier transform (IDFT) of the signal is taken. The chromatic dispersion in the TD is given as<disp-formula id="e9">
<mml:math id="m13">
<mml:mi>h</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x003D;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mi>D</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msqrt>
<mml:mo>.</mml:mo>
</mml:math>
<label>(9)</label>
</disp-formula>
</p>
<p>In the frequency domain, chromatic dispersion will be<disp-formula id="e10">
<mml:math id="m14">
<mml:mi>H</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x003D;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>j</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msup>
<mml:mo>&#x003D;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:math>
<label>(10)</label>
</disp-formula>where <italic>&#x3bb;</italic> is the light&#x2019;s wavelength, <italic>D</italic> is the dispersion coefficient of the optical fiber, <italic>&#x3b2;</italic>
<sub>2</sub> is the group velocity dispersion coefficient, <italic>&#x3c9;</italic> is the angular frequency, <italic>c</italic> is the speed of light, and <italic>L</italic>
<sub>
<italic>o</italic>
</sub> is the optical fiber length.</p>
<p>To obtain the numerical results, consider D &#x003D; 17&#xa0;<italic>ps</italic>/<italic>nm</italic> &#x2212; <italic>km</italic>, <italic>c</italic> &#x003D; 3 &#xd7; 10<sup>8</sup>&#xa0;m/s, and <italic>P</italic>
<sub>
<italic>in</italic>
</sub> &#x003D; 2 dBm. <xref ref-type="table" rid="T2">Table 2</xref> shows the parameters of the NN-based receiver used for the chromatic dispersion-based channel. The model is trained using Keras. During training, the number of Epochs &#x003D; 50. The loss function known as sparse categorical cross-entropy and optimizer known as Adam [<xref ref-type="bibr" rid="B36">36</xref>] are used during the training.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>NN receiver parameters for the chromatic dispersion-based channel.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Input signal</th>
<th align="center" colspan="4">Receiver</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center" rowspan="3">16-QAM</td>
<td align="center">Layers</td>
<td align="center">1</td>
<td align="center">2&#x2013;3</td>
<td align="center">4</td>
</tr>
<tr>
<td align="center">Neurons</td>
<td align="center">2</td>
<td align="center">50</td>
<td align="center">16</td>
</tr>
<tr>
<td align="center">Activation function</td>
<td align="center">ELU</td>
<td align="center">ELU</td>
<td align="center">ELU</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Chromatic dispersion broadens the signal due to which ISI occurs for different optical fiber lengths, <italic>L</italic>
<sub>
<italic>o</italic>
</sub>, as shown in <xref ref-type="fig" rid="F8">Figure 8</xref>. The effect of chromatic dispersion when <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 100&#xa0;km is shown in <xref ref-type="fig" rid="F8">Figure 8A</xref>. The received 16-QAM constellation is spread out, and the symbol error rate will increase. When fiber length decreases, the effect of chromatic dispersion on the signal is reduced, as shown for <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 75&#xa0;km in <xref ref-type="fig" rid="F8">Figure 8B</xref>. The signal constellation is less spread out, and ISI is lower. Therefore, the symbol error rate will reduce as the fiber length is reduced. <xref ref-type="fig" rid="F8">Figures 8C</xref> and <xref ref-type="fig" rid="F8">8D</xref> represent the effect of chromatic dispersion when the propagation distance is further reduced to <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 50&#xa0;km and <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 20&#xa0;km, respectively. It can be seen that the received constellation is slightly affected by the effect of chromatic dispersion. Hence, the spreading of the signal constellation is reduced further, which results in a reduction of the symbol error rate, as compared to <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 100&#xa0;km and <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 75&#xa0;km. To conclude, ISI is reduced by decreasing the fiber length in a chromatic dispersion-based channel, resulting in a reduction of the symbol error rate.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Effect of the chromatic dispersion-based channel for optical fiber lengths. <bold>(A)</bold> <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 100&#xa0;km. <bold>(B)</bold> <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 75&#xa0;km. <bold>(C)</bold> <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 50&#xa0;km. <bold>(D)</bold> <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 20&#xa0;km.</p>
</caption>
<graphic xlink:href="fphy-12-1387284-g008.tif"/>
</fig>
</sec>
</sec>
<sec id="s5">
<title>5 Split-step Fourier method</title>
<p>SSFM is a technique used for simultaneous simulation of self-phase modulation (SPM) and CD in optical fibers. First, nonlinearity alone is applied to the signal in the time domain. After that, the signal is converted into the frequency domain, and CD is applied. Therefore, <italic>&#x2202;</italic>/<italic>&#x2202;t</italic> in the NLSE is replaced with <italic>i&#x3c9;</italic>. The signal is converted back into the time domain by taking IDFT. The steps involved in this process can be described as follows [<xref ref-type="bibr" rid="B42">42</xref>]:<disp-formula id="e11">
<mml:math id="m15">
<mml:mi>X</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x002B;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x003D;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mfenced close="]" open="[">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mi>D</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mi>F</mml:mi>
<mml:mfenced close="]" open="[">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mi>X</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:math>
<label>(11)</label>
</disp-formula>where &#x2018;k&#x2019; is the small incremental distance and &#x2018;F&#x2019; is the fast Fourier transform.</p>
<sec id="s5-1">
<title>5.1 Iterative and symmetric SSFM</title>
<p>In this technique, dispersion is applied to the signal through half of the distance &#x2018;k&#x2019;, then nonlinearity acts on the middle of the distance, and finally, dispersion acts again on the remaining half of the distance. The operation is shown as [<xref ref-type="bibr" rid="B43">43</xref>]<disp-formula id="e12">
<mml:math id="m16">
<mml:mi>X</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x002B;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x003D;</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mi>exp</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x002B;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x002B;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
<mml:mi>exp</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mi>X</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:math>
<label>(12)</label>
</disp-formula>
</p>
<p>Several iterations have to be performed by considering an amplified optical communication system. For each span of the amplifier, dispersion and nonlinearity are mitigated, which leads to high complexity.</p>
</sec>
<sec id="s5-2">
<title>5.2 Noniterative asymmetric SSFM</title>
<p>To simplify iterative and symmetric SSFM, an assumption is made that nonlinearity acts only at the beginning of every amplifier. So, for each span, only one iteration is performed. Due to this, the complexity of the system is reduced significantly for the mitigation of nonlinear phase noise [<xref ref-type="bibr" rid="B44">44</xref>]. However, the performance of noniterative asymmetric SSFM is 2&#x2013;3&#xa0;dB poor compared to iterative SSFM. SSFM has a high computational cost of approximately 105 multiplications per symbol per channel. On the other side, it is a quite powerful simulation technique for mitigating both dispersion and nonlinearity [<xref ref-type="bibr" rid="B45">45</xref>].</p>
</sec>
<sec id="s5-3">
<title>5.3 Numerical results and discussions</title>
<p>For our numerical simulations, we consider iterative and symmetric split-step Fourier method under different fiber lengths. Consider <italic>&#x3b2;</italic>
<sub>2</sub> &#x003D; &#x2212;20 &#xd7; 10<sup>&#x2212;24</sup>&#xa0;<italic>s</italic>
<sup>2</sup>/<italic>km</italic>, <italic>&#x3b1;</italic> &#x003D; 0.2&#xa0;dB/km, <italic>&#x3b3;</italic> &#x003D; 1.27&#xa0;W/km, <italic>h</italic> &#x003D; 1&#xa0;km, and <italic>P</italic>
<sub>
<italic>in</italic>
</sub> &#x003D; 1 dBm. Here, <italic>&#x3b2;</italic>
<sub>2</sub> represents the GVD coefficient, nonlinearity coefficient is represented by <italic>&#x3b3;</italic>, <italic>h</italic> is the step size, <italic>P</italic>
<sub>
<italic>in</italic>
</sub> is the input power, and <italic>&#x3b1;</italic> is the attenuation. The neural network-based receiver parameters are shown in <xref ref-type="table" rid="T3">Table 3</xref>. The model is trained using Keras with number of Epochs being 120. A loss function known as sparse categorical cross-entropy is used. Adam optimizer [<xref ref-type="bibr" rid="B36">36</xref>] is used for training of the model. The results of different optical fiber lengths, <italic>L</italic>
<sub>
<italic>o</italic>
</sub>, used in the SSFM-based channel with 16-QAM input, and their decision regions formed are shown in <xref ref-type="fig" rid="F9">Figure 9</xref>.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>NN receiver parameters for the SSFM-based channel.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Input signal</th>
<th align="center" colspan="4">Receiver</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center" rowspan="3">16-QAM</td>
<td align="center">Layers</td>
<td align="center">1</td>
<td align="center">2&#x2013;5</td>
<td align="center">6</td>
</tr>
<tr>
<td align="center">Neurons</td>
<td align="center">2</td>
<td align="center">50</td>
<td align="center">16</td>
</tr>
<tr>
<td align="center">Activation function</td>
<td align="center">ELU</td>
<td align="center">ELU</td>
<td align="center">ELU</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Effect of the split-step Fourier method-based channel for different optical fiber lengths. <bold>(A)</bold> <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 100&#xa0;km. <bold>(B)</bold> <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 50&#xa0;km. <bold>(C)</bold> <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 25&#xa0;km. <bold>(D)</bold> <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 5&#xa0;km.</p>
</caption>
<graphic xlink:href="fphy-12-1387284-g009.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="F9">Figure 9A</xref> shows the effect of the SSFM-based channel for <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 100&#xa0;km. A small portion of the received symbols are intermixed with each other; hence, the symbol error rate will be high. For <xref ref-type="fig" rid="F9">Figure 9B</xref>, the received constellations for <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 50&#xa0;km are forming a boundary with each other, and there is little interference between the received symbols. In <xref ref-type="fig" rid="F9">Figure 9C</xref>, the length is reduced to <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 25&#xa0;km. For this length, the received constellation symbols do not interfere with each other and are some distance apart. <xref ref-type="fig" rid="F9">Figure 9D</xref> shows that for <italic>L</italic>
<sub>
<italic>o</italic>
</sub> &#x003D; 5&#xa0;km, the symbol error rate is minimum compared to other lengths, and received symbols are far apart from each other. Therefore, for short propagation distances, the channel effect on the received constellation is small and the symbol error rate is reduced.</p>
</sec>
</sec>
<sec id="s6">
<title>6 Conclusion and future work</title>
<p>The realm of short-reach optical communication has attracted significant attention both within academic circles and industry. Our focus has been on utilizing deep learning models to minimize symbol error rates in these types of optical communication setups. Various channel impairments, such as nonlinearity, CD, and attenuation, need to be accurately modeled. Initially, we addressed the challenge of modeling a nonlinear channel. Although conventional methods exist to tackle this issue, they tend to be intricate. Consequently, we harnessed a deep learning model called autoencoders to facilitate communication over nonlinear channels. This approach enabled the creation of an end-to-end system, yielding promising constellations. Furthermore, we investigated how the inclusion of a nonlinear channel within an autoencoder influences the received constellation as the optical fiber length increases. Another facet of our work involved the deployment of a deep neural network-based receiver utilizing a channel influenced by chromatic dispersion. By gradually extending the optical length, we explored its impact on the received constellation and, consequently, the symbol error rate. Finally, we incorporated the SSFM to emulate the combined effects of nonlinearities, chromatic dispersion, and attenuation in the optical channel. This was accomplished through a neural network-based receiver. The outcome was an evaluation of the symbol error rate as the optical fiber&#x2019;s length was augmented. Notably, we observed that the symbol error rate increases with the propagation distance of the optical fiber.</p>
<p>The potential for expanding upon this research lies in the application of various machine learning models to further reduce symbol error rates in short-reach optical communication. Although we have employed autoencoders and deep neural network-based receivers (decoders) in our current work, it is worth noting that the channel itself was not modeled using neural networks. To compute gradients in backpropagation, our abovementioned techniques always require a specific channel model. The precise mathematical relationship between the input and output of a real fiber channel is thus unknown, making these approaches inappropriate for real fiber channels. To overcome this limitation, an avenue worth exploring is the integration of reinforcement learning. This approach could enable the optimization of both the transmitter and receiver components independently, without requiring detailed channel knowledge. By leveraging reinforcement learning techniques, we could potentially enhance the overall performance of the system and achieve better symbol error rate outcomes in short-reach optical communication scenarios.</p>
</sec>
</body>
<back>
<sec id="s7" sec-type="data-availability">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s8">
<title>Author contributions</title>
<p>MI: software and writing&#x2013;original draft. AA: conceptualization and writing&#x2013;review and editing. AA: writing&#x2013;review and editing. JM: writing&#x2013;review and editing. IA: validation and writing&#x2013;review and editing. SG: methodology, project administration, validation, and writing&#x2013;review and editing. LP: conceptualization, methodology, and writing&#x2013;review and editing.</p>
</sec>
<sec id="s9" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research, authorship, and/or publication of this article.</p>
</sec>
<sec id="s10" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chagnon</surname>
<given-names>M</given-names>
</name>
</person-group>. <article-title>Optical communications for short reach</article-title>. <source>J Light Tech Nol</source> (<year>2019</year>) <volume>37</volume>:<fpage>1779</fpage>&#x2013;<lpage>97</lpage>. <pub-id pub-id-type="doi">10.1109/JLT.2019.2901201</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chaudhary</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>X</given-names>
</name>
</person-group>. <article-title>20Gbps MDM-based optical multimode system for short-haul communication</article-title>. In: <conf-name>Proceedings of the 2nd International Conference on Algorithms, Computing and Systems</conf-name>; <conf-date>July 27-29, 2018</conf-date>; <conf-loc>Beijing, China</conf-loc> (<year>2018</year>). <pub-id pub-id-type="doi">10.1145/3242840.3242885</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kani</surname>
<given-names>J-i.</given-names>
</name>
<name>
<surname>Terada</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Hatano</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>S-Y</given-names>
</name>
<name>
<surname>Asaka</surname>
<given-names>K</given-names>
</name>
<name>
<surname>Yamada</surname>
<given-names>T</given-names>
</name>
</person-group> <article-title>Future optical access network enabled by modularization and softwarization of access and transmission functions [Invited]</article-title>. <source>J Opt Commun Netw</source> (<year>2020</year>) <volume>12</volume>:<fpage>D48</fpage>&#x2013;<lpage>D56</lpage>. <pub-id pub-id-type="doi">10.1364/jocn.391544</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Satish Kumar</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Rajan</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Sushank</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Faisel</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Raad</surname>
<given-names>R</given-names>
</name>
</person-group>. <article-title>Developing cost-effective and high-speed 40 gbps FSO systems incorporating wavelength and spatial diversity techniques</article-title>. <source>Front Phys</source> (<year>2021</year>) <volume>9</volume>. <pub-id pub-id-type="doi">10.3389/fphy.2021.744160</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhong</surname>
<given-names>K</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Huo</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Lau</surname>
<given-names>APT</given-names>
</name>
</person-group> <article-title>Digital signal processing for short-reach optical communications: a review of current technologies and future trends</article-title>. <source>J Light Technol</source> (<year>2018</year>) <volume>36</volume>:<fpage>377</fpage>&#x2013;<lpage>400</lpage>. <pub-id pub-id-type="doi">10.1109/jlt.2018.2793881</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Karanov</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Chagnon</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Aref</surname>
<given-names>V</given-names>
</name>
<name>
<surname>Ferreira</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Lavery</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Bayvel</surname>
<given-names>P</given-names>
</name>
<etal/>
</person-group> <article-title>Experimental investigation of deep learning for digital signal processing in short reach optical fiber communications</article-title>. In: <conf-name>2020 IEEE Workshop on Signal Processing Systems (SiPS)</conf-name>; <conf-date>20-22 October 2020</conf-date>; <conf-loc>Coimbra, Portugal</conf-loc>. <publisher-name>IEEE</publisher-name> (<year>2020</year>). p. <fpage>1</fpage>&#x2013;<lpage>6</lpage>.</citation>
</ref>
<ref id="B7">
<label>7.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Che</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Bo</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>H</given-names>
</name>
</person-group> <article-title>Reduced-state mlse for an im/dd system using pam modulation</article-title>. <source>Opt Express</source> (<year>2020</year>) <volume>28</volume>:<fpage>38505</fpage>&#x2013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.1364/oe.410674</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Kalla</surname>
<given-names>SCK</given-names>
</name>
<name>
<surname>Rusch</surname>
<given-names>LA</given-names>
</name>
</person-group> <article-title>Recurrent neural nets achieving mlse performance in bandlimited optical channels</article-title>. In: <conf-name>2020 IEEE Photonics Conference (IPC)</conf-name>; <conf-date>10-14 November 2024</conf-date>; <conf-loc>Rome, Italy</conf-loc>. <publisher-name>IEEE</publisher-name> (<year>2020</year>). p. <fpage>1</fpage>&#x2013;<lpage>2</lpage>.</citation>
</ref>
<ref id="B9">
<label>9.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Gaiarin</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Pang</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Ozolins</surname>
<given-names>O</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>RT</given-names>
</name>
<name>
<surname>Da Silva</surname>
<given-names>EP</given-names>
</name>
<name>
<surname>Schatz</surname>
<given-names>R</given-names>
</name>
<etal/>
</person-group> <article-title>High speed pam-8 optical interconnects with digital equalization based on neural network</article-title>. In: <conf-name>2016 Asia Communications and Photonics Conference (ACP)</conf-name>; <conf-date>2-5 November 2016</conf-date>; <conf-loc>Wuhan, China</conf-loc>. <publisher-name>IEEE</publisher-name> (<year>2016</year>). p. <fpage>1</fpage>&#x2013;<lpage>3</lpage>.</citation>
</ref>
<ref id="B10">
<label>10.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ge</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>C</given-names>
</name>
<name>
<surname>He</surname>
<given-names>Z</given-names>
</name>
</person-group> <article-title>Compressed neural network equalization based on iterative pruning algorithm for 112-gbps vcsel-enabled optical interconnects</article-title>. <source>J Light Technol</source> (<year>2020</year>) <volume>38</volume>:<fpage>1323</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1109/jlt.2020.2973718</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Katz</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Sadot</surname>
<given-names>D</given-names>
</name>
</person-group> <article-title>Radial basis function network equalizer for optical communication ook system</article-title>. <source>J Lightwave Technology</source> (<year>2007</year>) <volume>25</volume>:<fpage>2631</fpage>&#x2013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1109/jlt.2007.902109</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Da Ros</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Ranzini</surname>
<given-names>SM</given-names>
</name>
<name>
<surname>Dischler</surname>
<given-names>R</given-names>
</name>
<name>
<surname>Cem</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Aref</surname>
<given-names>V</given-names>
</name>
<name>
<surname>B&#xfc;low</surname>
<given-names>H</given-names>
</name>
<etal/>
</person-group> <article-title>Machine-learning-based equalization for short-reach transmission: neural networks and reservoir computing</article-title>. In: <source>Proceedings Volume 1171205, Metro and Data Center Optical Networks and Short-Reach Links IV</source>. <publisher-name>SPIE</publisher-name> (<year>2021</year>). <pub-id pub-id-type="doi">10.1117/12.2583011</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Argyris</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Bueno</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Fischer</surname>
<given-names>I</given-names>
</name>
</person-group> <article-title>Photonic machine learning imple-mentation for signal recovery in optical communications</article-title>. <source>Sci Reports</source> (<year>2018</year>) <volume>8</volume>:<fpage>8487</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-018-26927-y</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ranzini</surname>
<given-names>SM</given-names>
</name>
<name>
<surname>Dischler</surname>
<given-names>R</given-names>
</name>
<name>
<surname>Da Ros</surname>
<given-names>F</given-names>
</name>
<name>
<surname>B&#xfc;low</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Zibar</surname>
<given-names>D</given-names>
</name>
</person-group> <article-title>Experi-mental investigation of optoelectronic receiver with reservoir computing in short reach optical fiber communications</article-title>. <source>J Light Technol</source> (<year>2021</year>) <volume>39</volume>:<fpage>2460</fpage>&#x2013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1109/jlt.2021.3049473</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Katumba</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Yin</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Dambre</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Bienstman</surname>
<given-names>P</given-names>
</name>
</person-group> <article-title>A neuromorphic silicon photonics nonlinear equalizer for optical communications with intensity modulation and direct detection</article-title>. <source>J Light Technol</source> (<year>2019</year>) <volume>37</volume>:<fpage>2232</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1109/jlt.2019.2900568</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Da Ros</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Ranzini</surname>
<given-names>SM</given-names>
</name>
<name>
<surname>Buelow</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Zibar</surname>
<given-names>D</given-names>
</name>
</person-group> <article-title>Reservoir-computing based equalization with optical pre-processing for short-reach optical transmission</article-title>. <source>IEEE J Sel Top Quan Electron.</source> (<year>2020</year>) <volume>26</volume>:<fpage>1</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1109/jstqe.2020.2975607</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Lyu</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>X</given-names>
</name>
</person-group> <article-title>Reservoir computing based equalization for radio over fiber system</article-title>. In: <conf-name>2021 23rd International Conference on Advanced Communication Technology (ICACT)</conf-name>; <conf-date>7-10 February 2021, PyeongChang</conf-date>; <conf-loc>South Korea</conf-loc>. <publisher-name>IEEE</publisher-name> (<year>2021</year>). p. <fpage>85</fpage>&#x2013;<lpage>90</lpage>.</citation>
</ref>
<ref id="B18">
<label>18.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luko&#x161;evi&#x2c7;cius</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Jaeger</surname>
<given-names>H</given-names>
</name>
</person-group>. <article-title>Reservoir computing approaches to recurrent neural network training</article-title>. <source>Comput Science Review</source> (<year>2009</year>) <volume>3</volume>(<issue>3</issue>):<fpage>127</fpage>&#x2013;<lpage>49</lpage>. <pub-id pub-id-type="doi">10.1016/j.cosrev.2009.03.005</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deligiannidis</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Bogris</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Mesaritakis</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Kopsinis</surname>
<given-names>Y</given-names>
</name>
</person-group> <article-title>Compen-sation of fiber nonlinearities in digital coherent systems leveraging long short-term memory neural networks</article-title>. <source>J Light Technol</source> (<year>2020</year>) <volume>38</volume>:<fpage>5991</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1109/jlt.2020.3007919</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Lavania</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Kumam</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Matey</surname>
<given-names>PS</given-names>
</name>
<name>
<surname>Annepu</surname>
<given-names>V</given-names>
</name>
<name>
<surname>Bagadi</surname>
<given-names>K</given-names>
</name>
</person-group> <article-title>Adap-tive channel equalization using recurrent neural network under sui channel model</article-title>. In: <conf-name>2015 International Conference on Innovations in In- formation, Embedded and Communication Systems (ICIIECS)</conf-name>; <conf-date>19-20 March 2015</conf-date>; <conf-loc>Coimbatore, India</conf-loc>. <publisher-name>IEEE</publisher-name> (<year>2015</year>). p. <fpage>1</fpage>&#x2013;<lpage>6</lpage>.</citation>
</ref>
<ref id="B21">
<label>21.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karanov</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Chagnon</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Thouin</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Eriksson</surname>
<given-names>TA</given-names>
</name>
<name>
<surname>B&#xfc;low</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Lavery</surname>
<given-names>D</given-names>
</name>
<etal/>
</person-group> <article-title>End-to-end deep learning of optical fiber communications</article-title>. <source>J Light Technol</source> (<year>2018</year>) <volume>36</volume>:<fpage>4843</fpage>&#x2013;<lpage>55</lpage>. <pub-id pub-id-type="doi">10.1109/jlt.2018.2865109</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>Q</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>M</given-names>
</name>
</person-group> <article-title>End-to- end learning for optical fiber communication with data-driven channel model</article-title>. In: <conf-name>2020 Opto-Electronics and Communications Conference (OECC)</conf-name>; <conf-date>4-8 October 2020</conf-date>; <conf-loc>Taipei, Taiwan</conf-loc>. <publisher-name>IEEE</publisher-name> (<year>2020</year>). p. <fpage>1</fpage>&#x2013;<lpage>3</lpage>.</citation>
</ref>
<ref id="B23">
<label>23.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Talreja</surname>
<given-names>V</given-names>
</name>
<name>
<surname>Koike-Akino</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Millar</surname>
<given-names>DS</given-names>
</name>
<name>
<surname>Kojima</surname>
<given-names>K</given-names>
</name>
<name>
<surname>Parsons</surname>
<given-names>K</given-names>
</name>
</person-group> <article-title>End-to-end deep learning for phase noise-robust multi-dimensional geometric shaping</article-title>. In: <conf-name>2020 European Conference on Opti- cal Communications (ECOC)</conf-name>; <conf-date>6-10 December 2020</conf-date>; <conf-loc>Brussels, Belgium</conf-loc>. <publisher-name>IEEE</publisher-name> (<year>2020</year>). p. <fpage>1</fpage>&#x2013;<lpage>3</lpage>.</citation>
</ref>
<ref id="B24">
<label>24.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Karanov</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Chagnon</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Aref</surname>
<given-names>V</given-names>
</name>
<name>
<surname>Lavery</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Bayvel</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Schmalen</surname>
<given-names>L</given-names>
</name>
</person-group> <article-title>Optical fiber communication systems based on end-to- end deep learning</article-title>. In: <conf-name>2020 IEEE Photonics Conference (IPC)</conf-name>; <conf-date>10-14 November 2024</conf-date>; <conf-loc>Rome, Italy</conf-loc>. <publisher-name>IEEE</publisher-name> (<year>2020</year>). p. <fpage>1</fpage>&#x2013;<lpage>2</lpage>.</citation>
</ref>
<ref id="B25">
<label>25.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>O&#x2019;Shea</surname>
<given-names>TJ</given-names>
</name>
<name>
<surname>Karra</surname>
<given-names>K</given-names>
</name>
<name>
<surname>Clancy</surname>
<given-names>TC</given-names>
</name>
</person-group> <article-title>Learning to communicate: channel auto-encoders, domain specific regularizers, and attention</article-title>. In: <conf-name>2016 IEEE International Symposium on Signal Processing and Information Technology (ISSPIT)</conf-name>; <conf-date>December 12-14, 2016</conf-date>; <conf-loc>Limassol, Cyprus</conf-loc>. <publisher-name>IEEE</publisher-name> (<year>2016</year>). p. <fpage>223</fpage>&#x2013;<lpage>8</lpage>.</citation>
</ref>
<ref id="B26">
<label>26.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>O&#x2019;shea</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Hoydis</surname>
<given-names>J</given-names>
</name>
</person-group> <article-title>An introduction to deep learning for the physical layer</article-title>. <source>IEEE Trans Cogn. Commun. Netw.</source> (<year>2017</year>) <volume>3</volume>:<fpage>563</fpage>&#x2013;<lpage>75</lpage>. <pub-id pub-id-type="doi">10.1109/tccn.2017.2758370</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>D&#xf6;rner</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Cammerer</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Hoydis</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Ten Brink</surname>
<given-names>S</given-names>
</name>
</person-group> <article-title>Deep learning based communication over the air</article-title>. <source>IEEE J Sel Top Signal Process</source> (<year>2017</year>) <volume>12</volume>:<fpage>132</fpage>&#x2013;<lpage>43</lpage>. <pub-id pub-id-type="doi">10.1109/jstsp.2017.2784180</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hadi</surname>
<given-names>MU</given-names>
</name>
<name>
<surname>Awais</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Raza</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Khurshid</surname>
<given-names>K</given-names>
</name>
<name>
<surname>Jung</surname>
<given-names>H</given-names>
</name>
</person-group>. <article-title>Neural Network DPD for Aggrandizing SM-VCSEL-SSMF-Based Radio over Fiber Link Performance</article-title>. <source>Photonics</source> (<year>2021</year>), <volume>8</volume>:<fpage>19</fpage>. <pub-id pub-id-type="doi">10.3390/photonics8010019</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Schaedler</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Kuschnerov</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Calabr&#x2018;o</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Pittal&#x2018;a</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Bluemm</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Pachnicke</surname>
<given-names>S</given-names>
</name>
</person-group> <article-title>Ai-based digital predistortion for iq mach-zehnder modulators</article-title>. In: <conf-name>2019 Asia Communications and Photonics Conference (ACP)</conf-name>; <conf-date>2-5 November 2019</conf-date>; <conf-loc>Chengdu, China</conf-loc>. <publisher-name>IEEE</publisher-name> (<year>2019</year>). p. <fpage>1</fpage>&#x2013;<lpage>3</lpage>.</citation>
</ref>
<ref id="B30">
<label>30.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sch&#xe4;dler</surname>
<given-names>M</given-names>
</name>
<name>
<surname>B&#xf6;cherer</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Pachnicke</surname>
<given-names>S</given-names>
</name>
</person-group> <article-title>Soft-demapping for short reach optical communication: a comparison of deep neural networks and volterra series</article-title>. <source>J Light Technol</source> (<year>2021</year>) <volume>39</volume>:<fpage>3095</fpage>&#x2013;<lpage>105</lpage>. <pub-id pub-id-type="doi">10.1109/jlt.2021.3056869</pub-id>
</citation>
</ref>
<ref id="B31">
<label>31.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schaedler</surname>
<given-names>M</given-names>
</name>
<name>
<surname>B&#xf6;cherer</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Pittal&#xe0;</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Calabr&#xf2;</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Stojanovic</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Bluemm</surname>
<given-names>C</given-names>
</name>
<etal/>
</person-group> <article-title>Recurrent neural network soft-demapping for nonlinear isi in 800gbit/s dwdm coherent optical transmissions</article-title>. <source>J Light Technol</source> (<year>2021</year>) <volume>39</volume>(<issue>16</issue>):<fpage>5278</fpage>&#x2013;<lpage>86</lpage>. <pub-id pub-id-type="doi">10.1109/JLT.2021.3102064</pub-id>
</citation>
</ref>
<ref id="B32">
<label>32.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Schaedler</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Calabr&#x2018;o</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Pittal&#x2018;a</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Bluemm</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Kuschnerov</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Pachnicke</surname>
<given-names>S</given-names>
</name>
</person-group> <article-title>Neural network-based soft-demapping for nonlinear channels</article-title>. In: <conf-name>Optical Fiber Communication Conference. Optical Society of America</conf-name>; <conf-date>8&#x2013;12 March 2020</conf-date>; <conf-loc>San Diego, California, United States</conf-loc> (<year>2020</year>). p. <fpage>W3D</fpage>&#x2013;<lpage>2</lpage>. <pub-id pub-id-type="doi">10.1364/ofc.2020.w3d.2</pub-id>
</citation>
</ref>
<ref id="B33">
<label>33.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Goodfellow</surname>
<given-names>I</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Courville</surname>
<given-names>A</given-names>
</name>
</person-group> <source>Deep learning</source>. <publisher-loc>Massachusetts, United States</publisher-loc>: <publisher-name>MIT press</publisher-name> (<year>2016</year>).</citation>
</ref>
<ref id="B34">
<label>34.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>LeCun</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>G</given-names>
</name>
</person-group> <article-title>Deep learning</article-title>. <source>nature</source> (<year>2015</year>) <volume>521</volume>(<issue>7553</issue>):<fpage>436</fpage>&#x2013;<lpage>44</lpage>. <comment>69</comment>. <pub-id pub-id-type="doi">10.1038/nature14539</pub-id>
</citation>
</ref>
<ref id="B35">
<label>35.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Nair</surname>
<given-names>V</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>GE</given-names>
</name>
</person-group> <article-title>Rectified linear units improve restricted boltz-mann machines</article-title>. In: <conf-name>Proceedings of the 27th international conference on machine learning (ICML-10)</conf-name>; <conf-date>June 21-24, 2010</conf-date>; <conf-loc>Haifa, Israel</conf-loc> (<year>2010</year>). p. <fpage>807</fpage>&#x2013;<lpage>14</lpage>.</citation>
</ref>
<ref id="B36">
<label>36.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kingma</surname>
<given-names>DP</given-names>
</name>
<name>
<surname>Ba</surname>
<given-names>J</given-names>
</name>
</person-group> <article-title>Adam: a method for stochastic optimization</article-title> (<year>2014</year>).</citation>
</ref>
<ref id="B37">
<label>37.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chollet</surname>
<given-names>F</given-names>
</name>
<etal/>
</person-group> <source>Keras</source> (<year>2015</year>).</citation>
</ref>
<ref id="B38">
<label>38.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>S</given-names>
</name>
<name>
<surname>H&#xe4;ger</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Garcia</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Wymeersch</surname>
<given-names>H</given-names>
</name>
</person-group> <article-title>Achievable information rates for nonlinear fiber communication via end-to-end autoencoder learning</article-title>. In: <conf-name>2018 European Conference on Optical Communication (ECOC)</conf-name>; <conf-date>September 23-27, 2018</conf-date>; <conf-loc>Rome, Italy</conf-loc> (<year>2018</year>). p. <fpage>1</fpage>&#x2013;<lpage>3</lpage>. <pub-id pub-id-type="doi">10.1109/ecoc.2018.8535456</pub-id>
</citation>
</ref>
<ref id="B39">
<label>39.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Agrawal</surname>
<given-names>G</given-names>
</name>
</person-group> <source>Fiber-optic communication systems</source>. <edition>6</edition>. <publisher-loc>new york</publisher-loc>: <publisher-name>John willey &#x26; son</publisher-name> (<year>2002</year>).</citation>
</ref>
<ref id="B40">
<label>40.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Keykhosravi</surname>
<given-names>K</given-names>
</name>
<name>
<surname>Durisi</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Agrell</surname>
<given-names>E</given-names>
</name>
</person-group> <article-title>A tighter upper bound on the capacity of the nondispersive optical fiber channel</article-title>. In: <conf-name>2017 European Conference on Optical Communication (ECOC)</conf-name>; <conf-date>17-21 September 2017</conf-date>; <conf-loc>Gothenburg, Sweden</conf-loc>. <publisher-name>IEEE</publisher-name> (<year>2017</year>). p. <fpage>1</fpage>&#x2013;<lpage>3</lpage>.</citation>
</ref>
<ref id="B41">
<label>41.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ho</surname>
<given-names>K-P</given-names>
</name>
</person-group> <source>Phase-modulated optical communication systems</source>. <publisher-loc>Cham</publisher-loc>: <publisher-name>Springer Science &#x26; Business Media</publisher-name> (<year>2005</year>).</citation>
</ref>
<ref id="B42">
<label>42.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kylemark</surname>
<given-names>P</given-names>
</name>
</person-group> <source>Nonlinear fiber optical technologies for transmission and amplification</source> (<year>2004</year>).</citation>
</ref>
<ref id="B43">
<label>43.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Agrawal</surname>
<given-names>GP</given-names>
</name>
</person-group> <article-title>Nonlinear fiber optics</article-title>. In: <source>Nonlinear science at the dawn of the 21st century</source>. <publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name> (<year>2000</year>). p. <fpage>195</fpage>&#x2013;<lpage>211</lpage>.</citation>
</ref>
<ref id="B44">
<label>44.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ip</surname>
<given-names>E</given-names>
</name>
<name>
<surname>Kahn</surname>
<given-names>JM</given-names>
</name>
</person-group> <article-title>Compensation of dispersion and nonlinear impairments using digital backpropagation</article-title>. <source>J Light Technol</source> (<year>2008</year>) <volume>26</volume>:<fpage>3416</fpage>&#x2013;<lpage>25</lpage>. <pub-id pub-id-type="doi">10.1109/jlt.2008.927791</pub-id>
</citation>
</ref>
<ref id="B45">
<label>45.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ip</surname>
<given-names>E</given-names>
</name>
<name>
<surname>Kahn</surname>
<given-names>J</given-names>
</name>
</person-group> <article-title>Increasing optical fiber transmission capacity beyond next-generation systems</article-title>. In: <conf-name>LEOS 2008-21st Annual Meeting of the IEEE Lasers and Electro-Optics Society</conf-name>; <conf-date>9-13 November 2008</conf-date>; <conf-loc>Newport Beach, USA</conf-loc> (<year>2008</year>). <pub-id pub-id-type="doi">10.1109/leos.2008.4688764</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>