<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Cardiovasc. Med.</journal-id>
<journal-title>Frontiers in Cardiovascular Medicine</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Cardiovasc. Med.</abbrev-journal-title>
<issn pub-type="epub">2297-055X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fcvm.2022.1067760</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Cardiovascular Medicine</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Myocardial strain analysis of echocardiography based on deep learning</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Deng</surname> <given-names>Yinlong</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1842799/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Cai</surname> <given-names>Peiwei</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2020;</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhang</surname> <given-names>Li</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Cao</surname> <given-names>Xiongcheng</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2100809/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Chen</surname> <given-names>Yequn</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1386084/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Jiang</surname> <given-names>Shiyan</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Zhuang</surname> <given-names>Zhemin</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x002A;</sup></xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Wang</surname> <given-names>Bin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2127760/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Cardiology, The First Affiliated Hospital of Shantou University Medical College</institution>, <addr-line>Shantou</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Preventive Medicine, Shantou University Medical College</institution>, <addr-line>Shantou</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Ultrasound Division, The First Affiliated Hospital of Shantou University Medical College</institution>, <addr-line>Shantou</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>Department of Electronic Information Engineering, College of Engineering, Shantou University</institution>, <addr-line>Shantou</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Guang Yang, Imperial College London, United Kingdom</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Carlos Francisco Moreno-Garcia, Robert Gordon University, United Kingdom; Zhonghua Sun, Curtin University, Australia; Heye Zhang, Sun Yat-sen University, China</p></fn>
<corresp id="c001">&#x002A;Correspondence: Bin Wang, <email>asch369@126.com</email></corresp>
<corresp id="c002">Zhemin Zhuang, <email>zmzhuang@stu.edu.cn</email></corresp>
<fn fn-type="equal" id="fn002"><p><sup>&#x2020;</sup>These authors have contributed equally to this work and share first authorship</p></fn>
<fn fn-type="other" id="fn004"><p>This article was submitted to Cardiovascular Imaging, a section of the journal Frontiers in Cardiovascular Medicine</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>16</day>
<month>12</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>9</volume>
<elocation-id>1067760</elocation-id>
<history>
<date date-type="received">
<day>12</day>
<month>10</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>30</day>
<month>11</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2022 Deng, Cai, Zhang, Cao, Chen, Jiang, Zhuang and Wang.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Deng, Cai, Zhang, Cao, Chen, Jiang, Zhuang and Wang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>Strain analysis provides more thorough spatiotemporal signatures for myocardial contraction, which is helpful for early detection of cardiac insufficiency. The use of deep learning (DL) to automatically measure myocardial strain from echocardiogram videos has garnered recent attention. However, the development of key techniques including segmentation and motion estimation remains a challenge. In this work, we developed a novel DL-based framework for myocardial segmentation and motion estimation to generate strain measures from echocardiogram videos.</p>
</sec>
<sec>
<title>Methods</title>
<p>Three-dimensional (3D) Convolutional Neural Network (CNN) was developed for myocardial segmentation and optical flow network for motion estimation. The segmentation network was used to define the region of interest (ROI), and the optical flow network was used to estimate the pixel motion in the ROI. We performed a model architecture search to identify the optimal base architecture for motion estimation. The final workflow design and associated hyperparameters are the result of a careful implementation. In addition, we compared the DL model with a traditional speck tracking algorithm on an independent, external clinical data. Each video was double-blind measured by an ultrasound expert and a DL expert using speck tracking echocardiography (STE) and DL method, respectively.</p>
</sec>
<sec>
<title>Results</title>
<p>The DL method successfully performed automatic segmentation, motion estimation, and global longitudinal strain (GLS) measurements in all examinations. The 3D segmentation has better spatio-temporal smoothness, average dice correlation reaches 0.82, and the effect of target frame is better than that of previous 2D networks. The best motion estimation network achieved an average end-point error of 0.05 &#x00B1; 0.03 mm per frame, better than previously reported state-of-the-art. The DL method showed no significant difference relative to the traditional method in GLS measurement, Spearman correlation of 0.90 (<italic>p</italic> &#x003C; 0.001) and mean bias &#x2212;1.2 &#x00B1; 1.5%.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>In conclusion, our method exhibits better segmentation and motion estimation performance and demonstrates the feasibility of DL method for automatic strain analysis. The DL approach helps reduce time consumption and human effort, which holds great promise for translational research and precision medicine efforts.</p>
</sec>
</abstract>
<kwd-group>
<kwd>echocardiography</kwd>
<kwd>strain</kwd>
<kwd>deep learning</kwd>
<kwd>segmentation</kwd>
<kwd>motion estimation</kwd>
</kwd-group>
<contract-num rid="cn001">2020LKSFG04B</contract-num>
<contract-num rid="cn002">2020B1515120061</contract-num>
<contract-sponsor id="cn001">Li Ka Shing Foundation<named-content content-type="fundref-id">10.13039/100007421</named-content></contract-sponsor>
<contract-sponsor id="cn002">Basic and Applied Basic Research Foundation of Guangdong Province<named-content content-type="fundref-id">10.13039/501100021171</named-content></contract-sponsor>
<counts>
<fig-count count="8"/>
<table-count count="3"/>
<equation-count count="5"/>
<ref-count count="39"/>
<page-count count="15"/>
<word-count count="9342"/>
</counts>
</article-meta>
</front>
<body>
<sec id="S1" sec-type="intro">
<title>1 Introduction</title>
<p>Assessment of cardiac mechanics has an essential role in diagnosis, risk stratification, and treatment strategies in patients with cardiac disease (<xref ref-type="bibr" rid="B1">1</xref>). The left ventricular ejection fraction (LVEF) is often used as a cardiac functional index, but it has a significant limitation when mechanical impairment occurs without an ejection fraction reduction (<xref ref-type="bibr" rid="B2">2</xref>). Alternatively, clinicians recommend finer markers of cardiac mechanical dysfunction (<xref ref-type="bibr" rid="B3">3</xref>). Strain imaging is richer description tool of cardiac function, which provides a more thorough characterization of myocardial contraction mechanics (<xref ref-type="bibr" rid="B4">4</xref>). It has applications in various cardiac pathologies (<xref ref-type="bibr" rid="B5">5</xref>, <xref ref-type="bibr" rid="B6">6</xref>). By assessing myocardial deformation, it can detect left ventricular dysfunction before a change in LVEF.</p>
<p>Currently, with rapid image acquisition and relatively low cost, the most extensively utilized modality in strain imaging is two-dimensional (2D) transthoracic speck tracking echocardiography (STE) (<xref ref-type="bibr" rid="B7">7</xref>), which is used to estimate pixel block motion within regions along the myocardial wall (<xref ref-type="bibr" rid="B8">8</xref>). However, it still has several unsolved challenges due to fundamental limitations of ultrasound image modality and algorithm ad-hoc setups, including inaccurate reflection of underlying biomechanical motion and some degree of non-conclusive results caused by errors related to image quality and algorithm assumptions (<xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B10">10</xref>). In addition, in clinical applications, there are several steps that require observer manual intervention such as view selection, adjustment of myocardial wall boundaries, and selection of tuneable parameters for tracking algorithms. This is a time-consuming and artificially introduced difference process that requires considerable expertise (<xref ref-type="bibr" rid="B10">10</xref>); the time spent completing a single global longitudinal strain (GLS) analysis has been found to range between 5 and 10 min (<xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B12">12</xref>), making it inefficient in clinical practice. As a result, it is difficult to integrate into the existing cardiac ultrasound workflow, limiting the wider clinical applicability of these techniques.</p>
<p>Artificial intelligence (AI) advances have opened up new possibilities in this regard. Many AI-based echocardiography interpretations have been proposed in recent years, some of which are equivalent to replicating what clinicians do with a visual diagnostic rather than a more thorough analysis of what is happening with the image (<xref ref-type="bibr" rid="B13">13</xref>). These methods are still limited in exploring the value of ultrasound images. Meticulous quantitative evaluation is another advantage of AI, such as pixel-level segmentation and motion prediction. Automatic delineation approaches have been implemented within computational pipelines (<xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B14">14</xref>, <xref ref-type="bibr" rid="B15">15</xref>). Recent studies have shown that motion tracking also can be treated as a learnable problem and is more robust than traditional approaches (e.g., variational) in some applications (<xref ref-type="bibr" rid="B16">16</xref>&#x2013;<xref ref-type="bibr" rid="B18">18</xref>). This has sparked a lot of interest in applying deep learning (DL) techniques to assess cardiac strain in echocardiography. &#x00D8;stvik et al. (<xref ref-type="bibr" rid="B19">19</xref>) have integrated cardiac view classification, event detection, myocardial segmentation, and motion estimation to construct an AI pipeline for fully automated GLS calculations. The Pwc-net (<xref ref-type="bibr" rid="B20">20</xref>) was used to learn to estimate ultrasonic motion, which performance was on par or better than state-of-the-art methods for traditional estimation. AI-pipeline has been shown as a promising alternative to traditional methods with significantly higher analytical efficiency. It is reported that the traditional method takes 5&#x2013;10 min to measure each time, while the AI method was performed in &#x003C;15 s (<xref ref-type="bibr" rid="B21">21</xref>). However, this method involves several sources of limited accuracy, especially the 2D segmentation and motion networks being the fundamental building blocks of the measurements.</p>
<p>Therefore, to improve clinical availability, we developed a more reliable DL framework for segmentation and motion estimation of echocardiogram videos to complete strain analysis. Segmentation, as a pre-step in motion estimation, is used to define the region of interest (ROI) and initialize the myocardial coordinate system for strain analysis. Previous attempts to segment left ventricular myocardium (LVM) with 2D U-net relied on manually labeled still images at end-systole (ES) and end-diastole (ED) instead of using the actual echocardiogram videos (<xref ref-type="bibr" rid="B22">22</xref>). These models are still limited in their ability to generalize, especially in the case of poor image quality (<xref ref-type="bibr" rid="B23">23</xref>). We used video instead of still frame trained three-dimensional (2D+t) CNN to segment myocardium. The three-dimensional (3D) convolution can simultaneously scan multiple frames and learn the relationships between them to constrain myocardial boundaries. Further, the optical flow CNN was employed to estimate pixel motion, which is the task of estimating the instantaneous velocity of pixels in the ROI (<xref ref-type="bibr" rid="B24">24</xref>). In previous studies, optical flow CNN for cardiac motion estimation from echocardiography has been demonstrated to be feasible, like EchoPwc-net and Flownet (<xref ref-type="bibr" rid="B19">19</xref>, <xref ref-type="bibr" rid="B25">25</xref>). But the best systems are still limited by difficulties including fast movement, occlusions, and motion blur. To address these deficiencies, we attempted a new motion estimation method and compared it with Pwc-net and Flownet. RAFT is a new motion estimation model proposed in recent years (<xref ref-type="bibr" rid="B24">24</xref>). The algorithm uses additional recurrent neural network to optimize motion details, and has better performance in fast motion and occlusion. In our work, we proposed a 3D network for echocardiographic myocardial segmentation and compared several classical motion estimation algorithms to achieve global and local strain analysis.</p>
<p>Therefore, in this study, we constructed a new fast, fully automatic myocardial strain analysis workflow consisting of segmentation and motion estimation convolutional neural networks. This approach provides visually assessable tracings of the myocardial motion, which facilitate human assessment and downstream analysis. And that, the accuracy and repeatability of the proposed framework are verified <italic>in vivo</italic> data, which is critical for clinical adoption (<xref ref-type="bibr" rid="B10">10</xref>). Compared with previous studies, our model has more potential in automatic quantitative analysis of echocardiography. This could make the measurements more robust and hopefully replace traditional methods to achieve automatic strain measurement without observer intervention, helping to improve clinical workflow.</p>
</sec>
<sec id="S2" sec-type="materials|methods">
<title>2 Materials and methods</title>
<sec id="S2.SS1">
<title>2.1 The architecture of deep learning model</title>
<p>We construct a DL workflow for cardiac segmentation, motion estimation, and calculating GLS, which has three key components (<xref ref-type="fig" rid="F1">Figure 1</xref>). Let <italic>I<sub>t</sub></italic> be a frame at time <italic>t</italic>.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption><p>Overview of the deep learning (DL) workflow for automatic measurement of longitudinal strain. 3D-CSN uses multi-frame echocardiogram images as input to delineate the region of interest and segments, and extract the centerline of target frame. Then, the motion network estimates the movement field of the pixels in the ROI to generate the velocity vector at each moment. These velocities are used to propagate the position of the centerline and then calculate the global and local line arc lengths coexisting in {Zk} which are used as a basis for strain measurements.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcvm-09-1067760-g001.tif"/>
</fig>
</sec>
<sec id="S2.SS2">
<title>2.2 Segmentation and LVM localization</title>
<p>First, we developed a 3D cardiac segmentation network (3D-CSN) with U-net type for frame-level semantic segmentation of the cardiac to generate ROI and establish <italic>I<sub>0</sub></italic> LVM coordinate system for strain analysis. The segmentation network is a 3D architecture that uses a 3D array with size 256 &#x00D7; 256 &#x00D7; n (256 &#x00D7; 256 represents the image size, and n represent the number of frames) to generate a segmentation mask of equal size, each color corresponding to a tissue label (i.e., background and LVM). It has an encoder and a decoder path each with four layers. In the encoder path, each layer contains two 3D convolutions (3 &#x00D7; 3 &#x00D7; 3) each followed by a rectified linear unit (ReLu) and max pooling (2 &#x00D7; 2 &#x00D7; 2). In the decoder path, each layer consists of an upconvolution (2 &#x00D7; 2 &#x00D7; 2) followed by two 3D convolutions each followed by a ReLu. The kernel of 3D convolution allows the model to simultaneously scan three successive frames and learn spatial features in two dimensions (x, y) as well as temporal information in the third dimension (<italic>t</italic>), which has previously performed well in video classification tasks (<xref ref-type="bibr" rid="B26">26</xref>). The encoder path is used to learn video spatiotemporal features, and then generate the same resolution mask through decoder path. Skip connection is used to propagate details. Thus, 3D-CSN uses a 3-channel input volume composed of 3 consecutive frames of images to generate a 3-channel array o of equal size, each channel representing the mask of each frame. The target image and mask are overlapped to obtain the segmentation result, namely the ROI (&#x03A9;). Skeleton extraction algorithm is then employed to take the centerline (<sub>&#x0297;</sub>&#x2286;&#x03A9;) between the endocardium and the epicardium on end-diastole (<italic>I<sub>0</sub></italic>), which will be used to locate the myocardium and generate coordinates. The centerline, defined as the mid-point between two nearest endo- and epicardial points, was used to track and calculate myocardial strain.</p>
</sec>
<sec id="S2.SS3">
<title>2.3 Motion estimation</title>
<p>Second, we constructed an optical-flow based motion estimation network used to estimate each pixel (v&#x2208;&#x03A9;) movement field (<italic>f<sub>t</sub></italic>) of the heart from <italic>I<sub>t</sub></italic> to <italic>I</italic><sub><italic>t+1</italic></sub>, which is used to update the position of LVM centerline (&#x0297;). In this study, we employ three different variants (RAFT, Pwc-net, and Flownet) to identify the optimum basic architecture and eventually chose RAFT as the best performing architecture. RAFT, based on an optimization approach, consists of three main components: feature encoder, correlation layer, and update operator. This method involves taking two consecutive images (<italic>I<sub>t</sub></italic>, <italic>I</italic><sub><italic>t+1</italic></sub>) as input, and these are fed separately into feature encoders with shared weights. The feature encoder contains six residual blocks and is downsampled three times; the amount of filters successively is 64,64,128,128, and 192,192. After feature extraction, the block maps the input images to dense feature maps (<italic>g</italic>&#x03B8;) at 1/8 resolution. Then we compute visual similarity by constructing the dot product between all pairs of feature vectors. It can be efficiently computed as single matrix multiplication</p>
<disp-formula id="S2.E1">
<label>(1)</label>
<mml:math id="M1">
<mml:mrow>
<mml:mrow>
<mml:mpadded width="+3.3pt">
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mpadded>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:munder>
<mml:mo largeop="true" movablelimits="false" symmetric="true">&#x2211;</mml:mo>
<mml:mi>h</mml:mi>
</mml:munder>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi mathvariant="normal">&#x03B8;</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>.</mml:mo>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi mathvariant="normal">&#x03B8;</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <italic>C</italic><sub><italic>ijkl</italic></sub> is the correlation layer, <italic>ij</italic> and <italic>kl</italic> represent pixel coordinates of <italic>g</italic>&#x03B8;(<italic>I</italic><sub>1</sub>) and <italic>g</italic>&#x03B8;(<italic>I</italic><sub>2</sub>), respectively, and h denotes the channel of feature maps.</p>
<p>The correlation volume <italic>C</italic><sub><italic>ijkl</italic></sub> contains four correlation layers of different sizes by pooling <italic>kl</italic> dimensions with kernel sizes 1, 2, 4, and 8 strides, which stored both large and small displacements information. Meanwhile, it maintains high-resolution information by maintaining the <italic>ij</italic> dimensions, allowing this method could recover the motions of small fast-moving objects. Finally, the network uses the current flow field <italic>f<sub>k</sub></italic> to retrieve correlation features <italic>C</italic> from the correlation layer and then concatenate them to input ConvGRU-based (<xref ref-type="bibr" rid="B27">27</xref>) update operator that produces an update direction &#x0394;<italic>f</italic> to update <italic>f<sub>k</sub></italic>, <italic>f<sub>k</sub></italic> is initially set to zero (<bold>f<sub>0</sub></bold> = <bold>0</bold>). Update operator is a lightweight network based on recurrent neural network, which can set any number of iterations to optimize the flow field. After 12 iterations, we decided on the final optical flow. Thus, RAFT uses two consecutive 1-channel input images with size 256 &#x00D7; 256 to generate a 2-channel array <italic>f</italic> of equal size, each channel representing the x and y components of motion, which is used to update the x and y coordinates of LVM pixels, respectively. Structural comparisons of RAFT, Pwc-net, and Flownet are discussed in <xref ref-type="supplementary-material" rid="DS1">Supplementary section 1</xref>.</p>
</sec>
<sec id="S2.SS4">
<title>2.4 Tracking update and calculate strain</title>
<p>Finally, we update the position of myocardial centerline <sub>&#x0297;</sub> by displacement field <italic>f</italic>, i.e., <sub>&#x0297;</sub>(<italic>t</italic> + 1) = <sub>&#x0297;</sub>(<italic>t</italic>)+ <italic>f</italic>(<italic>t</italic>), and calculate its arclength &#x03C4; which represents the longitudinal length of LVM at each moment. Strain represents percent change in myocardial fibers length per unit under stress (<xref ref-type="bibr" rid="B13">13</xref>). Thus &#x03C4; is used to estimate GLS</p>
<disp-formula id="S2.E2">
<label>(2)</label>
<mml:math id="M2">
<mml:mrow>
<mml:mrow>
<mml:mi>&#x03C2;</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo rspace="5.8pt">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">&#x03C4;</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>-</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">&#x03C4;</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mi mathvariant="normal">&#x03C4;</mml:mi>
</mml:mrow>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where &#x03C4;(<italic>t</italic>) denotes ventricular longitudinal length at frame <italic>t</italic>, &#x03C4;(0) is ED frame ventricular longitudinal length. The peak-GLS was defined as the minima strain value within a cardiac cycle. For segments, we adopt 16-segment division method (<xref ref-type="fig" rid="F2">Figure 2</xref>), with the apex of ED centerline as the demarcation node, divided three arcs of the same size on both sides as the initial position of each segment (<xref ref-type="bibr" rid="B28">28</xref>), and calculated regional longitudinal strain (RLS) in subsequent updates.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption><p>Echocardiography 16-segment. The 2015 ASE guideline recommend that the left ventricular myocardium be divided into three rings, namely the basal ring, the mid ring, and the apical ring, each with a height 1/3 of the length of the left ventricle. The basal ring and the mid ring contain six segments, respectively, and the apical ring contains four segments. The actual echocardiogram corresponds to the diagram above. The colors indicate different supplying coronary arteries.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcvm-09-1067760-g002.tif"/>
</fig>
</sec>
</sec>
<sec id="S3">
<title>3 Experiments</title>
<sec id="S3.SS1">
<title>3.1 Data preparation</title>
<p>The echocardiography dataset consists of two parts, one for model training and the other for external clinical validation.</p>
<p>For development, we used a publicly available dataset (<xref ref-type="bibr" rid="B29">29</xref>) of simulated echocardiography images, consisting of 105 sequences or 6,165 frames with apical 2-chamber (A2C), apical 3-chamber (A3C), and apical 4-chamber (A4C) views. The data is created with a complex biomechanical model for comparison of speck tracking imaging algorithms. Data templates come from seven different vendors and five motion patterns, including one healthy and four pathologies. For each frame, the authors provide a set of 180 points coordinate <italic>P</italic><sub><italic>k</italic></sub> &#x003C;x, y&#x003E;, <italic>k</italic> {1,2&#x2026;180} evenly distributed in the myocardial region, corresponding to the underlying motion field of the LV myocardium. Therefore, semi-automatic labeling was employed to generate labels for monitoring network training. As shown in <xref ref-type="fig" rid="F3">Figure 3A</xref>, we annotate the region of LVM in each frame by a concave hull that connects the peripheral points, medial line is defined as the endocardium and the lateral line as the epicardium. Inside the myocardium labels &#x201C;1,&#x201D; others label &#x201C;0.&#x201D; Every three consecutive frames as a data is converted to a 3D array, with the corresponding label as input. In order to generalize the model in reality, we additionally fine-tuned the model on 200 manually annotated datasets. For motion estimation network, we used the coordinate changes of corresponding points between consecutive frames to generate a sparse displacement field, i.e., <italic>V<sub>k</sub> = P<sub>k</sub></italic> (t + 1) &#x2212; <italic>P<sub>k</sub></italic> (t), and then used cubic interpolation to convert it to a dense displacement map with velocities inside the myocardium as shown in <xref ref-type="fig" rid="F3">Figure 3B</xref>. The dense Displacement map is used as the ground truth flow to monitor gradient descent of the motion network. In order to expand the range of motion amplitude distribution between frames, we used sampling every other frame to clip the raw video. Each video was cut into 4 cine-loop. Finally, 4200 fully labeled echocardiographic data were generated for 3D-CSN training, and 12,000 ultrasound image pairs for the development of the motion network. For the robustness of model, data augmentation including basic augmentations and ultrasound-specific augmentation routines was applied (<xref ref-type="bibr" rid="B19">19</xref>). Details of data edit and enhancement are provided in <xref ref-type="supplementary-material" rid="DS1">Supplementary section 2</xref>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption><p>Semi-automatic data annotation. <bold>(A)</bold> Segmentation marked, the concave hull formed by connecting the outermost point is marked as the LVM region. All pixels within the hull are marked as &#x201C;1,&#x201D; as shown in yellow, and those outside are marked as &#x201C;0,&#x201D; as shown in purple. Every three frames as a data. <bold>(B)</bold> Optical flow marked, <italic>I<sub>t</sub></italic>, <italic>I</italic><sub><italic>t+1</italic></sub> represent two adjacent frames. The sparse displacement field is calculated according to the position changes of the corresponding points of the before and after frames. Then, the density velocity field of each pixel is generated by cubic interpolation, which is represented as flow field. Color and saturation indicate direction and magnitude, respectively.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcvm-09-1067760-g003.tif"/>
</fig>
<p>Additionally, we evaluated the actual effectiveness of the model on an external test dataset of 150 echocardiogram videos from 50 patients from an independent hospital system (The First Affiliated Hospital of Shantou University Medical College) and in comparison with commerce STE. Three standard apical views were extracted from each patient. To ensure a representative range of LV pathologies, we included three pre-defined patient groups (<xref ref-type="supplementary-material" rid="DS1">Supplementary section 3</xref>): 17 patients with myocardial infarction, 17 patients with ischemic heart failure, and 16 patients admitted for chest pain without any evidence of cardiac origin. Videos were acquired by skilled sonographers using PHILIPS Epiq 7C ultrasound machine and processed images were stored in a Philips Xcelera picture archiving and communication system. The study was approved by the Regional Committee for Medical and Health Research Ethics (No. B-2022-196) and all patients gave written consent. Patients were included consecutively for each group regardless of image quality. Exclusion criteria were significant ventricular aneurysm, atrial fibrillation, age younger than 18 years, or inability to give written informed consent. <xref ref-type="supplementary-material" rid="DS1">Supplementary Table 1</xref> summarizes the characteristics of the included patients.</p>
</sec>
<sec id="S3.SS2">
<title>3.2 Model development and training</title>
<p>Model design and training were done in Python using the PyTorch deep learning library. The hardware consisted of an Intel Core CPU i7-11700k, 32 GB RAM, and an NVIDIA RTX3090 GPU with 24 GB of memory. The training process is divided into two sections: The 3D semantic segmentation network training and motion estimation network training. All data are randomly divided into training set, validation set and test set in 7:2:1 order. Model development adopts transfer learning strategy. Segmentation network was pre-trained on simulated ultrasonic data, then fine-tuned on hand-labeled data. Motion estimation network was pre-trained on FlyingThings, then fine-tuned on the abovementioned ultrasonic data.</p>
<p>The essence of segmentation is pixel classification, means that every pixel on the image needs to be assigned to the target region (<xref ref-type="bibr" rid="B30">30</xref>). In other words, the task of 3D-CSN is to classify every pixel on the video volume into LVM or background. Model input is a 256 &#x00D7; 256 &#x00D7; 3 voxel tile together with a binary value indicating classification of each pixel,256 &#x00D7; 256 represents the space size of the image, and 3 indicates the number of frames. The output is a mask of the same size as the input video, where each pixel is classified as either LVM or background. To enable training of 3D network we used the memory efficient cuDNN convolution layer implementation. The model was initialized with random weights and was trained using the stochastic gradient descent with momentum (SGDM) optimizer. We set batch size of 6 and learning rate of 2e-4 for all experiments. We ran 50k and 10k iterations on pre-trained and fine-tuned, respectively, which took approximately 25 h. Different from previous Dice loss in 2D segmentation, this network output and ground truth are compared using Dice loss + Cross entropy</p>
<disp-formula id="S3.E3">
<label>(3)</label>
<mml:math id="M3">
<mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>Y</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>Y</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>-</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:munderover>
<mml:mo largeop="true" movablelimits="false" symmetric="true">&#x2211;</mml:mo>
<mml:mrow>
<mml:mpadded width="+3.3pt">
<mml:mi>c</mml:mi>
</mml:mpadded>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>C</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:munderover>
<mml:mo largeop="true" movablelimits="false" symmetric="true">&#x2211;</mml:mo>
<mml:mrow>
<mml:mpadded width="+3.3pt">
<mml:mi>n</mml:mi>
</mml:mpadded>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>g</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mmultiscripts>
<mml:mi>y</mml:mi>
<mml:none/>
<mml:mo>&#x2032;</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:none/>
</mml:mmultiscripts>
</mml:msub>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mstyle scriptlevel="-1">
<mml:mo largeop="true" mathsize="160%" stretchy="false" symmetric="true">&#x22C2;</mml:mo>
</mml:mstyle>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2062;</mml:mo>
<mml:mmultiscripts>
<mml:mi>y</mml:mi>
<mml:none/>
<mml:mo>&#x2032;</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:none/>
</mml:mmultiscripts>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>|</mml:mo>
<mml:mo>+</mml:mo>
<mml:mo>|</mml:mo>
<mml:mi>y</mml:mi>
<mml:mmultiscripts>
<mml:mo>|</mml:mo>
<mml:mprescripts/>
<mml:none/>
<mml:mo>&#x2032;</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:none/>
</mml:mmultiscripts>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where, <italic>y</italic><sub><italic>n,c</italic></sub>&#x2208;<italic>Y</italic> and y&#x2032;<sub>n,c</sub>&#x2208;<italic>Y</italic>&#x2032; are the target label and predictionof class <italic>C</italic> and <italic>N</italic>th in batch processing, respectively, <italic>Y</italic> and <italic>Y</italic>&#x2032; are the truth value and prediction result of input image, and <italic>C</italic> and <italic>N</italic> represent the number of classes and pixels in the dataset. P represent probability.</p>
<p>The model-predicted and labels were compared using Dice Similarity Coefficient (DSC) metrics at ED, ES, and random other frames. We examine the model&#x2019;s segmentation effect on three views and compare it with the 2D segmentation model involved in the pipeline proposed by Smistad et al. (<xref ref-type="bibr" rid="B31">31</xref>). Note that 3D-CSN is trained on the data set we developed, while the 2D network is trained on CAMUS (<xref ref-type="bibr" rid="B32">32</xref>) data set (including the tagged ED and ES frames).</p>
<p>Optical flow network was initialized with pre-trained weights from the FlyingThings dataset, then fine-tuned on ultrasonic data. The model&#x2019;s input is the segmentation network&#x2019;s output, and the output is the dense displacement field of ultrasonic specks in two consecutive frames. We tested three different model architectures. All networks were trained with AdamW optimizer parameters beta 1, 2 = 0.9, 0.999, random initialization, and the initial learning rate = 2e-4. For fine-tuning, the initial learning rate was set to 1e-4. We pre-trained on FlyingThings for 100k iterations with a batch size of 12 and then fine-tuned on echocardiography for an additional 20k steps with a batch size of 6. Training time was approximately 2&#x2013;3 days. The L2 distance between the predicted and ground truth flow was used to supervise network training. The loss is defined as</p>
<disp-formula id="S3.E4">
<label>(4)</label>
<mml:math id="M4">
<mml:mrow>
<mml:mpadded width="+3.3pt">
<mml:mi>L</mml:mi>
</mml:mpadded>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:munderover>
<mml:mo largeop="true" movablelimits="false" symmetric="true">&#x2211;</mml:mo>
<mml:mrow>
<mml:mpadded width="+3.3pt">
<mml:mi>i</mml:mi>
</mml:mpadded>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="normal">&#x03B3;</mml:mi>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mo>-</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2062;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo fence="true">||</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>-</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo fence="true">||</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <italic>N</italic> represents iteration times, <italic>f</italic><sub><italic>gt</italic></sub> stand for flow ground truth, <italic>f<sub>i</sub></italic> represents the <italic>i</italic>th predicted flow, &#x03B3; = 0.8.</p>
<p>The accuracy was assessed using the in-plane end-point error (EPE) between predicted V&#x2019; and ground truth <italic>V</italic>, it is defined as the Euclidean distance between the ground truth velocity and the predictions.</p>
<disp-formula id="S3.E5">
<label>(5)</label>
<mml:math id="M5">
<mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:mi>E</mml:mi>
</mml:mpadded>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>v</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo rspace="5.8pt" stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:msup>
<mml:msqrt>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>x</mml:mi>
</mml:msub>
<mml:mo>-</mml:mo>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
<mml:mo>-</mml:mo>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Finally, our model tracked the position of the centerline frame by frame based on the optical flow field and used this as the longitudinal length of the myocardium. Then the strain value at each moment was calculated according to formula (2), and finally formed the myocardial strain curve.</p>
</sec>
<sec id="S3.SS3">
<title>3.3 Prospective clinical validation</title>
<p>A comparison study was performed by analyzing 150 echocardiogram videos to compare the proposed DL method and actual clinical practice measurement. We used this model and STE to test the exact same echocardiogram simultaneously, resulting in 2 paired GLS measurements and RLS value. GLS was calculated as the average peak strain of the 3 apical views. A single heart cycle (start at ED) was chosen from each view and the exact same recording and cardiac cycle was used for both methods. ED was defined by the automatic ECG trigger algorithm of the analysis software. The first measurement system consisted of a single experienced observer using a commercially available speck-tracking analyses method (Philips auto-CMQ) for strain measurements. The observer manually corrected the ROI by visual assessment of the endocardial and epicardial borders. Block matching was used for motion tracking. Spatial and temporal smoothing were kept at default values. Drift compensation was applied as by default. The speck tracking analyses were performed in accordance with the consensus document of the EACVI/ASE/Industry Task Force to standardize deformation imaging (<xref ref-type="bibr" rid="B28">28</xref>). The second measurement system was the DL method measuring strain without any observer intervention. The model automatically performs ROI identification, tissue division, motion estimation, and strain calculation. Analyses were performed without knowledge of clinical data or previous measurement results. Finally, to assess whether the consistency between the two methods was affected by LV pathologic type and image view, subgroup analysis of 150 echocardiographic videos was performed, classified by LV pathologic type (normal, myocardial infarction, and heart failure) and image view (A2C, A3C, or A4C).</p>
</sec>
<sec id="S3.SS4">
<title>3.4 Statistical analysis</title>
<p>Continuous variables were presented as mean &#x00B1; standard deviation, and dichotomous data were presented as numbers (percentages). Association between methods was estimated by calculating the Pearson correlation coefficient. Bland&#x2013;Altman (B-A) analysis and intra-group consistency comparison (ICC) was used to assess the agreement of measurement pairs. The mean absolute difference between the two measurement systems was calculated using the mean value of the absolute difference between all measurement pairs, and bias denotes the mean difference was calculated using the mean value of the difference between all measurement pairs. Tests for normality were performed using Shapiro&#x2013;Wilk and Kolmogorov&#x2013;Smirnov tests. ANOVA was used to assess if there was a statistically significant difference in bias between subgroups of measurement pairs when categorized using view and pathological pattern, or Brown-Forsythe test when the variance is not uniform. All statistical analyses were performed using SPSS 25.0 software (SPSS Inc, Chicago, IL, USA).</p>
</sec>
</sec>
<sec id="S4" sec-type="results">
<title>4 Results</title>
<sec id="S4.SS1">
<title>4.1 Model for segmentation</title>
<p>3D-CSN was trained to generate frame-level segmentations for the entire video (<xref ref-type="fig" rid="F4">Figure 4A</xref>). The DSC of LVM was measured and collected in <xref ref-type="table" rid="T1">Table 1</xref>. For comparison, DSC obtained with the 2D network proposed by Smistad et al. (<xref ref-type="bibr" rid="B31">31</xref>) is also included in the table. With this model, the correlation of 3D-CSN measures and LVM labels was &#x003E;0.79 across all measures, and the average DSC for A2C/A3C/A4C was 0.826/0.808/0.833, respectively. Further, the average DSC of 2D segmentation was 0.787/0.720/0.798. The 2D and 3D networks have achieved comparable results on ED and ES frames, but 2D network significantly worse than 3D-CSN (i.e., A3C: other 0.803 vs. 0.581) on other frames. The 3D model showed similar performance on each frame of the sequence, while the 2D model performed well on ED/ES but poorly on intermediate sequences (<xref ref-type="fig" rid="F4">Figure 4B</xref>). Both 3D and 2D models perform better on the A4C plane.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption><p>Segmentation performance. <bold>(A)</bold> 3D-CSN allows simultaneous segmentation of multiple frames. Inputting echocardiogram video sequence frames (top row) will generate corresponding segmentation masks (bottom row). <bold>(B)</bold> The dice similarity coefficient (DSC) was calculated for each frame of video on different models.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcvm-09-1067760-g004.tif"/>
</fig>
<table-wrap position="float" id="T1">
<label>TABLE 1</label>
<caption><p>State-of-the-art method for left-ventricular myocardial segmentation shown at end-diastole (ED), end-systole (ES), and one of other sequence frame compared to 3D-CSN.</p></caption>
<table cellspacing="5" cellpadding="5" frame="box" rules="all">
<thead>
<tr>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Method</td>
<td valign="top" align="center" colspan="3" style="color:#ffffff;background-color: #7f8080;">A2C Dice</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Avg.</td>
<td valign="top" align="center" colspan="3" style="color:#ffffff;background-color: #7f8080;">A3C Dice</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Avg.</td>
<td valign="top" align="center" colspan="3" style="color:#ffffff;background-color: #7f8080;">A4C Dice</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Avg.</td>
</tr>
<tr>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;"></td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">ED</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">ES</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Other</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;"></td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">ED</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">ES</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Other</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;"></td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">ED</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">ES</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Other</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;"></td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">3D-CSN</td>
<td valign="top" align="center">0.836</td>
<td valign="top" align="center">0.810</td>
<td valign="top" align="center">0.832</td>
<td valign="top" align="center">0.826</td>
<td valign="top" align="center">0.823</td>
<td valign="top" align="center">0.799</td>
<td valign="top" align="center">0.803</td>
<td valign="top" align="center">0.808</td>
<td valign="top" align="center">0.856</td>
<td valign="top" align="center">0.829</td>
<td valign="top" align="center">0.814</td>
<td valign="top" align="center">0.833</td>
</tr>
<tr>
<td valign="top" align="left">Smistad et al. (<xref ref-type="bibr" rid="B31">31</xref>)</td>
<td valign="top" align="center">0.802</td>
<td valign="top" align="center">0.857</td>
<td valign="top" align="center">0.703</td>
<td valign="top" align="center">0.787</td>
<td valign="top" align="center">0.786</td>
<td valign="top" align="center">0.794</td>
<td valign="top" align="center">0.581</td>
<td valign="top" align="center">0.720</td>
<td valign="top" align="center">0.811</td>
<td valign="top" align="center">0.839</td>
<td valign="top" align="center">0.694</td>
<td valign="top" align="center">0.798</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn><p>Avg., average.</p></fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="S4.SS2">
<title>4.2 Model for motion estimation</title>
<p>Five-fold cross-validation was performed on the simulated ultrasonic data, this resulted in five training sessions for each network. The average EPE (AEPE) with corresponding standard deviation can be seen in <xref ref-type="table" rid="T2">Table 2</xref>. The smallest AEPE session as the best model, RAFT/Pwc-net/Flownet is 0.05/0.08/0.12. RAFT had lower AEPE relative to Pwc-net and Flownet, and Pwc-net was better than Flownet (<italic>p</italic> &#x003C; 0.001). <xref ref-type="fig" rid="F5">Figure 5</xref> illustrates a representative example of dense flow results for the different methods and the end-point distance between ground-truth and prediction of displacement field.</p>
<table-wrap position="float" id="T2">
<label>TABLE 2</label>
<caption><p>Average end point error (AEPE) on simulated ultrasound data.</p></caption>
<table cellspacing="5" cellpadding="5" frame="box" rules="all">
<thead>
<tr>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Method</td>
<td valign="top" align="center" colspan="5" style="color:#ffffff;background-color: #7f8080;">AEPE [mm]</td>
</tr>
<tr>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;"></td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Fold 1</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Fold 2</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Fold 3</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Fold 4</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Fold 5</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">RAFT</td>
<td valign="top" align="center">0.05 &#x00B1; 0.04</td>
<td valign="top" align="center">0.06 &#x00B1; 0.03</td>
<td valign="top" align="center"><bold>0.05 &#x00B1; 0.03</bold></td>
<td valign="top" align="center">0.07 &#x00B1; 0.05</td>
<td valign="top" align="center">0.07 &#x00B1; 0.03</td>
</tr>
<tr>
<td valign="top" align="left">Pwc-net</td>
<td valign="top" align="center">0.10 &#x00B1; 0.07</td>
<td valign="top" align="center">0.09 &#x00B1; 0.05</td>
<td valign="top" align="center">0.12 &#x00B1; 0.09</td>
<td valign="top" align="center"><bold>0.08 &#x00B1; 0.06</bold></td>
<td valign="top" align="center">0.10 &#x00B1; 0.08</td>
</tr>
<tr>
<td valign="top" align="left">Flownet</td>
<td valign="top" align="center">0.15 &#x00B1; 0.10</td>
<td valign="top" align="center"><bold>0.12 &#x00B1; 0.08</bold></td>
<td valign="top" align="center">0.13 &#x00B1; 0.08</td>
<td valign="top" align="center">0.14 &#x00B1; 0.10</td>
<td valign="top" align="center">0.13 &#x00B1; 0.12</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn><p>Units given in mm per timestep/frame &#x0394;T &#x2013; 1. Data are presented as mean &#x00B1; standard deviation. Bold font are the best results for each model.</p></fn>
</table-wrap-foot>
</table-wrap>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption><p>Example of predicted optical flow patterns for the different models within the myocardium. The <bold>top row</bold> is the sequence of input to motion estimation networks. The <bold>middle row</bold> is the prediction of networks, represented by color coded with hue values, color and saturation indicate direction and magnitude, respectively. The <bold>bottom row</bold> shows the velocity vector comparison between ground truth and the different methods inside the locality as indicated by the dotted box in the ultrasound image. Light green arrows are ground truth, while red arrows are predictions.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcvm-09-1067760-g005.tif"/>
</fig>
</sec>
<sec id="S4.SS3">
<title>4.3 Tracking update and clinical validation</title>
<p>A comparison of tracking throughout the whole cardiac cycle from a simulated ultrasonic subject is shown in <xref ref-type="supplementary-material" rid="VS1">Supplementary Video 1</xref>, together with the formation of GLS curves. For Clinic validation, GLS was obtained successfully in all patients. A correlation plot of the peak-GLS measured by proposed method and reference method for each individual view and the average is given in <xref ref-type="fig" rid="F6">Figure 6</xref>, and correlations were 0.83, 0.87, 0.78, and 0.90, respectively. The average peak-GLS on all subjects was &#x2212;13.53 &#x00B1; 3.04% and &#x2212;14.72 &#x00B1; 3.39% for the DL method and reference method (<xref ref-type="table" rid="T3">Table 3</xref>). The mean absolute difference was 1.6 &#x00B1; 1.1%. The Bland-Altman analysis of between method differences revealed a bias of &#x2212;1.2 &#x00B1; 1.5% (<italic>p</italic> &#x003C; 0.01) with estimated limits of agreement (LOA) of &#x2212;4.1 to 1.7% (<xref ref-type="fig" rid="F7">Figure 7</xref>). There is no significant difference in bias between DL method and reference method among subgroups classified by view (<italic>p</italic> = 0.86). Moreover, no significant difference in bias was found between subgroups when categorized using pathological patterns (<italic>p</italic> = 0.07). The consistency analysis of measurement results differences between subgroups is presented in <xref ref-type="fig" rid="F8">Figures 8A,B</xref>. The intra-group consistency comparison (ICC) of RLS of the 16 segments contained in the 3 views is shown in <xref ref-type="supplementary-material" rid="DS1">Supplementary Table 2</xref>. RLS showed good consistency only in the basal anterolateral, mid anterolateral and apical anterior, with ICC exceeding 0.8, and large instability in other segments. The bull&#x2019;s eye plots display the RLS value of 16 segments measured by STE and DL method for different subjects (<xref ref-type="supplementary-material" rid="DS1">Supplementary Figure 3</xref>). In healthy subject, strain values in the polar map have a similar distribution. In MI patient, both maps indicate a focal strain reduction at the lower right, and inspection of the myocardium on the echocardiography shows an inferolateral infarct that coincides in location with segments with more prominent decreases in strain. In HF patient, both maps indicate a diffused strain reduction.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption><p>Correlation plot of global longitudinal strain (GLS) estimated between STE and deep learning-based method for specific views and averaged over the three apical views. Each dot represents one examination. Blue dotted line represents the best fit line to the data by linear regression. Green solid line represents the theoretical perfect correlation.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcvm-09-1067760-g006.tif"/>
</fig>
<table-wrap position="float" id="T3">
<label>TABLE 3</label>
<caption><p>Mean global longitudinal strain (GLS) for each view measured by deep learning method and reference method (standard deviation), and averaged over the three apical views.</p></caption>
<table cellspacing="5" cellpadding="5" frame="box" rules="all">
<thead>
<tr>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Method</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">A2C GLS</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">A3C GLS</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">A4C GLS</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Average</td>
</tr>
<tr>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">DL-method</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">&#x2212;14.06 &#x00B1; 3.36%</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">&#x2212;13.29 &#x00B1; 3.57%</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">&#x2212;13.24 &#x00B1; 3.28%</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">&#x2212;13.53 &#x00B1; 3.04%</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Reference-method</td>
<td valign="top" align="center">&#x2212;15.26 &#x00B1; 3.35%</td>
<td valign="top" align="center">&#x2212;14.43 &#x00B1; 3.81%</td>
<td valign="top" align="center">&#x2212;14.47 &#x00B1; 3.73%</td>
<td valign="top" align="center">&#x2212;14.72 &#x00B1; 3.39%</td>
</tr>
</tbody>
</table></table-wrap>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption><p>Bland&#x2013;Altman plot presents the comparison of measurements between the reference method and the DL method. The figure shows the limits of agreement (LOA) of 1.96 SD (red dotted line). SD, standard deviation.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcvm-09-1067760-g007.tif"/>
</fig>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption><p>Bland&#x2013;Altman plot presents the comparison of measurements between the reference method and the DL method. <bold>(A)</bold> Measurement pairs labeled by pathology, <bold>(B)</bold> and specific views. MI, myocardial infarction; IHF, ischemic heart failure.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcvm-09-1067760-g008.tif"/>
</fig>
<p>The computational timing of the proposed method is approximately 5 s per video for myocardial segmentation and 100 ms per frame for motion estimation on a GPU of a standard desktop computer. Total processing time when running the entire workflow was 13 &#x00B1; 2 s per view and 40 &#x00B1; 5 s for a full patient analysis including all three apical views. On the same echocardiography, STE analysis was completed within 3 min per video.</p>
</sec>
</sec>
<sec id="S5" sec-type="discussion">
<title>5 Discussion</title>
<p>Our study describes a carefully designed automatic strain quantification DL workflow that consists of 3D segmentation and optical flow network for handling the challenges associated with echocardiographic motion tracking. It was able to determine myocardial borders, estimate motion, and ultimately compute strain. Based on tracking movement of centerline initialized from myocardial segmentation at the first frame, the DL approach measured strain frame by frame. The tracking was performed by using the displacement fields from the motion estimation network to update the position of points on the centerline. We benchmarked its segmentation, motion, and strain estimation components against the state-of-the-art. We compared our segmentation and motion estimation to other DL methods, and strain measures to a reference speck tracking technique.</p>
<p>3D-CSN was designed to execute semantic segmentation for determining the ROI of motion estimation network and initializing LVM position. Noise is a long-standing difficulty in the field of motion estimation in echocardiography (<xref ref-type="bibr" rid="B33">33</xref>). We mask the ultrasound image to remove redundant input signals and determine the LVM wall boundary. A 2D segmentation network was utilized to determine the LVM wall in &#x00D8;stvik et al.&#x2019;s study (<xref ref-type="bibr" rid="B19">19</xref>), which was trained only on ED and ES frames but segmented each frame of echocardiogram video. Its segmentation is separate, ignoring the temporal and spatial continuity of myocardial movement. This result in model performed well on ED and ES frames but not on intermediate frames. In this investigation, we employed U-net with 3D convolution (3D-CSN) to complete this challenge. In all views, 3D-CSN exhibited a higher average dice score in segmenting regions and performed similarly well on all frames of a cardiac cycle. Both models were based on U-net framework, the main difference is 3D vs. 2D convolution. Two-dimensional convolution is a 3x3 block that can only scan one image at a time to learn spatial information of the image, whereas 3D convolution is a 3&#x00D7;3&#x00D7;3 volume that could scan multiple images simultaneously to learn the spatial features and inter-frame relationship of the image. Myocardial motion is regular, and when multiple frames are segmented at once, the border of the target frame will be restricted by the preceding and following frames, thus the 3D-CSN segmentation results are more spatio-temporal smoothness. In addition, DL is a typical data-driven model, the range of data is closely related to the generalization performance (<xref ref-type="bibr" rid="B4">4</xref>). The 3D-CSN is trained within all sequence frames, so its generalization ability is stronger. Furthermore, we believe that this design is more consistent with biological characteristics. In echocardiography, even experienced ultrasound experts sometimes have difficulty in accurately identifying the endocardium and epicardium in a single image frame, especially in the image with significant noise and artifacts, usually by repeatedly viewing the video. Therefore, 3D-CSN, similar to multi-frame observation, can infer the position of the endo- and epicardium of current frame from adjacent frames, which seems helpful for reducing the influence of image quality on segmentation model. To our knowledge, 3D-CSN is the first 3D U-net-based DL model for echocardiogram segmentation which could learn the morphological features of all frames and its performance better than that of previous image-based 2D networks. Motion estimation is another crucial part of quantifying myocardial deformation. We compared three different network structures. <xref ref-type="table" rid="T2">Table 2</xref> suggests that the motion estimation method producing best results is RAFT, and Pwc-net is superior to Flownet. The qualitative results in <xref ref-type="fig" rid="F5">Figure 5</xref> further suggest that RAFT has a better match in velocity vectors. In recent years, optical flow network has experienced the development from encoder-decoder architecture to spatial pyramid structure and then to optimization-based network. All three kinds of networks have outstanding performance in motion estimation, but which is most suitable for echocardiography has not been determined. &#x00D8;stvik et al. (<xref ref-type="bibr" rid="B19">19</xref>, <xref ref-type="bibr" rid="B25">25</xref>) have described the applications of Pwc-net and Flownet in myocardial movement, where Pwc-net outperforms Flownet, which was in line with our study due to Pwc-net introducing a refinement mechanism that optimizes flow prediction based on the pyramidal layers. However, recent reports suggest that spatial pyramid architecture may ignore small, fast-moving motions. Pwc-net adopts a coarse-to-fine strategy to refine the flow, that is, the optical flow is initialized at the minimum resolution first and then refined in the direction of high resolution, hence named pyramid architecture. This structure may result in fast-moving small objects being missed in low-resolution and challenging to recover in later iterations. Due to the lower displacement magnitudes between frames in echocardiography compared to other film actions (<xref ref-type="bibr" rid="B19">19</xref>), we question the validity of this architecture. In our study, we compared the RAFT and ultimately chose it as a component for motion estimation due to its excellent performance in minor motion and occlusion (<xref ref-type="bibr" rid="B34">34</xref>, <xref ref-type="bibr" rid="B35">35</xref>). In synthetic echocardiography, the AEPE of RAFT is significantly less than Pwc-net (<xref ref-type="table" rid="T2">Table 2</xref>). Unlike Pwc-net, RAFT eschews the pyramidal refinement structure, instead performing a update operator consisting of recurrent neural network to refine optical flow by generating an unlimited number of iterations at the same resolution, which could integrate last flow deviation to optimize the current flow prediction. We set up 12 iterations and did it at a single high resolution so that the slight motion of the first iteration could also be transmitted to the last layer. Similar to the human eye, people lack an intuitive understanding of some details of movement due to the limited field of view at the low resolution. Whereas RAFT is like multiple observations at the exact resolution, repeatedly reinforcing the details. By comparison, RAFT&#x2019;s specific architecture is more compliant with echocardiographic motion. We believe that further changes based on RAFT will become the new benchmark.</p>
<p>Therefore, our design is more in line with the biological characteristics of human experts. The LVM borders is identified by dynamic observation and eliminates the interference of noise from other regions to motion estimation. Then, the movement of each pixel was optimized for multiple iterations at a single high resolution instead of coarse-to-fine. In both the segmentation and motion estimation, our models are on par or better than the state-of-the-art.</p>
<p>The DL method successfully completed all GLS measurements and has a high level of agreement with currently accepted speck tracking techniques. Compared to semi-automatic cardiac motion quantification (CMQ) in Phillips ultrasound, the average GLS were &#x2212;13.53 &#x00B1; 3.04% and &#x2212;14.72 &#x00B1; 3.39%, respectively, showing an excellent correlation that achieved 0.90 (<italic>p</italic> &#x003C; 0.001), and consistency analysis exhibited bias of &#x2212;1.2% with LOA of &#x00B1;2.9%. A study by Salte et al. (<xref ref-type="bibr" rid="B21">21</xref>) compared the previously mentioned computing pipeline with traditional methods. They reported bias of &#x2212;1.4% and LOA of &#x00B1;3.7%. Our results are within that range. Assessing subgroups categorized by view and pathology, there was also no statistically significant difference between DL and reference methods, suggesting that different views or pathological motion states had limited influence on consistency. The strain measured by traditional STE is not the gold standard, so we cannot determine which method is more accurate, but it is still sufficient to demonstrate the accuracy and repeatability of DL automatic strain calculation. Nevertheless, there is a large variability in regional strain, which is expected because the two methods do not achieve uniform anatomical constraints, hence there are differences in the definition of segments. However, DL method is still sensitive to the changes of local strain under pathological conditions. As with STE, the bull&#x2019;s eye map (<xref ref-type="supplementary-material" rid="DS1">Supplementary Figure 3</xref>) from DL shows reduced strain that coincides with infarct segments in patients with infarction and reduced diffuse strain in patients with heart failure, suggesting that the DL approach is also diagnostic for myocardial disease. The currently most widely used semi-automatic speck tracking method is time-consuming and demands expertise. It involves several steps of operator intervention, such as view selection, ROI adjustment, and tunable parameter setting. Therefore, it is difficult to integrate into the existing workflow of ultrasonic cardiogram, which is mainly completed by offline analysis. The measurement accuracy of DL is comparable to or better than that of traditional methods, and the efficiency is significantly improved. With only one GPU, the DL method completes these tasks in real-time; each prediction task takes &#x003C;15 s and is much more rapid than the STE assessment of myocardial strain. The rapidity and automaticity of AI greatly decrease the labor of cardiac function assessment and experiential needs. This provides the opportunity for more-frequent, rapid evaluations of cardiac strain (e.g., on-screen view strain real-time during image acquisition and application in portable ultrasound). DL methods could potentially aid clinicians with a more precise and rapid assessment of cardiac strain and early detect abnormal ventricular wall movement. In settings in which the sensitive detection of change in cardiac function is critical, early detection of change can substantially affect clinical care (<xref ref-type="bibr" rid="B36">36</xref>, <xref ref-type="bibr" rid="B37">37</xref>).</p>
<p>It is worth noting that in our tests, we found that tracking may fail with poor image quality, as Salte et al. (<xref ref-type="bibr" rid="B21">21</xref>) reported. Poor image quality in echocardiography is a common problem that leads to invalid measurements. However, most previous DL studies have no effective monitoring means, and the primary limitation of AI in clinical application is that the internal mechanism is unclear. Hence, it is difficult for clinicians to trust it directly. Our method is different from the past&#x2019;s black-box approach proposed by some authors, such as directly predicting LVEF or GLS from images (<xref ref-type="bibr" rid="B38">38</xref>, <xref ref-type="bibr" rid="B39">39</xref>). AI models should be designed to provide visual feedback that can be checked manually. As a result, offering a visual feedback system is a vital assurance of the results&#x2019; trustworthiness. The method presented in this study was able to visually inspect if segmentation and motion estimates seem reasonable by visualizing myocardial segmentation and centerline movement (<xref ref-type="supplementary-material" rid="VS2">Supplementary Video 2</xref>). So when the tracking fails, it can be clearly sensed by the observer.</p>
<p>Our work provides a more accurate and robust scheme for automated GLS analysis and is the first model to realize local strain analysis on echocardiography. Compared with the previous strain analysis models, we propose improvements to segmentation and motion estimation. All strain measurements are successfully completed, and the results are better than previous advanced methods. This approach is expected to replace STE for real-time strain calculation in cardiac ultrasound practice. Thus, if these DL methods are integrated into a generic AI pipeline, the individual steps could be computed during the acquisition of images, allowing for rapid bedside analysis and even real-time measurements on the ultrasound scanner.</p>
<sec id="S5.SS1">
<title>5.1 Study limitations</title>
<p>This study of DL applied to echocardiographic data has several limitations. First, the learning-based optical flow method has high precision and efficient reasoning ability. However, obtaining training data in reality is difficult, so most supervised methods heavily rely on large-scale synthetic data sets. The training data used in this study came from the processing of data synthesized according to the biological template and we are not clear about the difference between this data and real <italic>in vivo</italic> data. Whether training data will lead model preference remains to be further investigation. Second, DL is a method based on data-driven, which does not have reasoning ability. It tends to represent the distribution of training data. Data regional differences lead to potential degradation when models are transferred to the real world. Further, as the template of simulated data is equal across categories, we also suspect that the motion model is slightly over-fitting. We believe further optimizations can be made including real-world data acquisition and self-supervised model training. Finally, for external validation, we only compared measurements against one commerce method and lacked the gold standard metric. Thus, we cannot conclude whether the method in this study is more accurate than the traditional STE method. We could only conclude that there is a high degree of agreement of GLS measurements between the two measurement systems, while the regional strain estimates still have large variability and need to be further optimized. Due to the limited sample size, the conclusions of comparative studies may not be generalizable. Therefore, this part of the work should be considered as a pilot study for clinical comparison, further studies with cardiac magnetic resonance as the gold standard and larger sample sizes should be carried out.</p>
</sec>
</sec>
<sec id="S6" sec-type="conclusion">
<title>6 Conclusion</title>
<p>Fully automated strain measurements based on DL have the potential to both reduce manual intervention and improve reproducibility, and, due to the processing speed of learning-based algorithms, this could eventually enable on-screen measurements in real-time while the operator acquires images. Our carefully designed structure is state-of-the-art, making the workflow an excellent candidate for use in routine clinical studies or data-driven research. In future studies, we will further optimize it to achieve robust multi-dimensional strain analysis, which will help to obtain diagnostic information more quickly and accurately, hopefully replacing traditional measurement methods to optimize clinical flow.</p>
</sec>
<sec id="S7" sec-type="data-availability">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="S8" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>The studies involving human participants were reviewed and approved by Ethics Committee of The First Affiliated Hospital of Shantou University Medical College. The patients/participants provided their written informed consent to participate in this study.</p>
</sec>
<sec id="S9" sec-type="author-contributions">
<title>Author contributions</title>
<p>BW and ZZ: conceptualization and design. YD: design the workflow, perform the data curation, and draft the manuscript. PC: image acquisition. LZ, YC, XC, and SJ: modify the manuscript. BW: project administration and funding acquisition. All authors revised the drafted manuscript, contributed to critical intellectual content, and read and agreed to the published version of the manuscript.</p>
</sec>
</body>
<back>
<sec id="S10" sec-type="funding-information">
<title>Funding</title>
<p>This work was supported, in part, by the Li Ka Shing Foundation (LKSF) under Grant 2020LKSFG04B and Basic and Applied Basic Research Foundation of Guangdong Province [2020B1515120061].</p>
</sec>
<ack><p>The authors wish to thank the enrolled patients for their participation and research staff for their effort in the study.</p>
</ack>
<sec id="S11" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="S12" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="S13" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fcvm.2022.1067760/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fcvm.2022.1067760/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.docx" id="DS1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Video_1.MP4" id="VS1" mimetype="video/mp4" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Video_2.MP4" id="VS2" mimetype="video/mp4" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<title>Abbreviations</title>
<fn fn-type="abbr">
<p>DL, deep learning; CNN, convolutional neural network; STE, speck-tracking echocardiography; ED, end-diastole; ES, end-systole; LVEF, left ventricular ejection fraction; EPE, end-point error; LVM, left-ventricular myocardium; 2D, two-dimensional; 3D, three-dimensional; GLS, global longitudinal strain; 3D-CSN, three-dimensional cardiac segmentation network; A2C, apical 2-chamber; A3C, apical 3-chamber; A4C, apical 4-chamber; DSC, dice similarity coefficient; AI, artificial intelligence; ROI, region of interest; RLS, regional longitudinal strain.</p></fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1"><label>1.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Members</surname> <given-names>A</given-names></name> <name><surname>McMurray</surname> <given-names>J</given-names></name> <name><surname>Adamopoulos</surname> <given-names>S</given-names></name> <name><surname>Anker</surname> <given-names>S</given-names></name> <name><surname>Auricchio</surname> <given-names>A</given-names></name> <name><surname>B&#x00F6;hm</surname> <given-names>M</given-names></name><etal/></person-group> <article-title>ESC Guidelines for the diagnosis and treatment of acute and chronic heart failure 2012: the task force for the diagnosis and treatment of acute and chronic heart failure 2012 of the European society of cardiology. Developed in collaboration with the heart failure association (HFA) of the ESC.</article-title> <source>Eur Heart J.</source> (<year>2012</year>) <volume>33</volume>:<fpage>1787</fpage>&#x2013;<lpage>847</lpage>.</citation></ref>
<ref id="B2"><label>2.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Konstam</surname> <given-names>M</given-names></name> <name><surname>Abboud</surname> <given-names>F</given-names></name></person-group>. <article-title>Ejection fraction: misunderstood and overrated (changing the paradigm in categorizing heart failure).</article-title> <source><italic>Circulation.</italic></source> (<year>2017</year>) <volume>135</volume>:<fpage>717</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1161/CIRCULATIONAHA.116.025795</pub-id> <pub-id pub-id-type="pmid">28223323</pub-id></citation></ref>
<ref id="B3"><label>3.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cikes</surname> <given-names>M</given-names></name> <name><surname>Solomon</surname> <given-names>S</given-names></name></person-group>. <article-title>Beyond ejection fraction: an integrative approach for assessment of cardiac structure and function in heart failure.</article-title> <source><italic>Eur Heart J.</italic></source> (<year>2016</year>) <volume>37</volume>:<fpage>1642</fpage>&#x2013;<lpage>50</lpage>. <pub-id pub-id-type="doi">10.1093/eurheartj/ehv510</pub-id> <pub-id pub-id-type="pmid">26417058</pub-id></citation></ref>
<ref id="B4"><label>4.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Duchateau</surname> <given-names>N</given-names></name> <name><surname>King</surname> <given-names>A</given-names></name> <name><surname>De Craene</surname> <given-names>M</given-names></name></person-group>. <article-title>Machine learning approaches for myocardial motion and deformation analysis.</article-title> <source><italic>Front Cardiovasc Med.</italic></source> (<year>2020</year>) <volume>6</volume>:<issue>190</issue>. <pub-id pub-id-type="doi">10.3389/fcvm.2019.00190</pub-id> <pub-id pub-id-type="pmid">31998756</pub-id></citation></ref>
<ref id="B5"><label>5.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nesbitt</surname> <given-names>G</given-names></name> <name><surname>Mankad</surname> <given-names>S</given-names></name> <name><surname>Oh</surname> <given-names>J</given-names></name></person-group>. <article-title>Strain imaging in echocardiography: methods and clinical applications.</article-title> <source><italic>Int J Cardiovasc Imaging.</italic></source> (<year>2009</year>) <volume>25</volume>:<fpage>9</fpage>&#x2013;<lpage>22</lpage>. <pub-id pub-id-type="doi">10.1007/s10554-008-9414-1</pub-id> <pub-id pub-id-type="pmid">19145475</pub-id></citation></ref>
<ref id="B6"><label>6.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>L</given-names></name> <name><surname>Huang</surname> <given-names>X</given-names></name> <name><surname>Ma</surname> <given-names>J</given-names></name> <name><surname>Huang</surname> <given-names>J</given-names></name> <name><surname>Fan</surname> <given-names>Y</given-names></name> <name><surname>Li</surname> <given-names>H</given-names></name><etal/></person-group> <article-title>Value of three-dimensional strain parameters for predicting left ventricular remodeling after ST-elevation myocardial infarction.</article-title> <source><italic>Int J Cardiovasc Imaging.</italic></source> (<year>2017</year>) <volume>33</volume>:<fpage>663</fpage>&#x2013;<lpage>73</lpage>. <pub-id pub-id-type="doi">10.1007/s10554-016-1053-3</pub-id> <pub-id pub-id-type="pmid">28150084</pub-id></citation></ref>
<ref id="B7"><label>7.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gorcsan</surname> <given-names>J</given-names></name> <name><surname>Tanaka</surname> <given-names>H</given-names></name></person-group>. <article-title>Echocardiographic assessment of myocardial strain.</article-title> <source><italic>J Am Coll Cardiol.</italic></source> (<year>2011</year>) <volume>58</volume>:<fpage>1401</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1016/j.jacc.2011.06.038</pub-id> <pub-id pub-id-type="pmid">21939821</pub-id></citation></ref>
<ref id="B8"><label>8.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nagel</surname> <given-names>E</given-names></name></person-group>. <article-title>Tissue tracking technology for assessing cardiac mechanics: principles, normal values, and clinical applications.</article-title> <source><italic>JACC Cardiovasc Imaging.</italic></source> (<year>2015</year>) <volume>8</volume>:<fpage>1444</fpage>&#x2013;<lpage>60</lpage>. <pub-id pub-id-type="doi">10.1016/j.jcmg.2015.11.001</pub-id> <pub-id pub-id-type="pmid">26699113</pub-id></citation></ref>
<ref id="B9"><label>9.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Morales</surname> <given-names>M</given-names></name> <name><surname>Van den Boomen</surname> <given-names>M</given-names></name> <name><surname>Nguyen</surname> <given-names>C</given-names></name> <name><surname>Kalpathy-Cramer</surname> <given-names>J</given-names></name> <name><surname>Rosen</surname> <given-names>B</given-names></name> <name><surname>Stultz</surname> <given-names>C</given-names></name><etal/></person-group> <article-title>DeepStrain: a deep learning workflow for the automated characterization of cardiac mechanics.</article-title> <source><italic>Front Cardiovasc Med.</italic></source> (<year>2021</year>) <volume>8</volume>:<issue>730316</issue>. <pub-id pub-id-type="doi">10.3389/fcvm.2021.730316</pub-id> <pub-id pub-id-type="pmid">34540923</pub-id></citation></ref>
<ref id="B10"><label>10.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Amzulescu</surname> <given-names>M</given-names></name> <name><surname>De Craene</surname> <given-names>M</given-names></name> <name><surname>Langet</surname> <given-names>H</given-names></name> <name><surname>Pasquet</surname> <given-names>A</given-names></name> <name><surname>Vancraeynest</surname> <given-names>D</given-names></name> <name><surname>Pouleur</surname> <given-names>A</given-names></name><etal/></person-group> <article-title>Myocardial strain imaging: review of general principles, validation, and sources of discrepancies.</article-title> <source><italic>Eur Heart J Cardiovasc Imaging.</italic></source> (<year>2019</year>) <volume>20</volume>:<fpage>605</fpage>&#x2013;<lpage>19</lpage>. <pub-id pub-id-type="doi">10.1093/ehjci/jez041</pub-id> <pub-id pub-id-type="pmid">30903139</pub-id></citation></ref>
<ref id="B11"><label>11.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Manovel</surname> <given-names>A</given-names></name> <name><surname>Dawson</surname> <given-names>D</given-names></name> <name><surname>Smith</surname> <given-names>B</given-names></name> <name><surname>Nihoyannopoulos</surname> <given-names>P</given-names></name></person-group>. <article-title>Assessment of left ventricular function by different speckle-tracking software.</article-title> <source><italic>Eur J Echocardiogr.</italic></source> (<year>2010</year>) <volume>11</volume>:<fpage>417</fpage>&#x2013;<lpage>21</lpage>. <pub-id pub-id-type="doi">10.1093/ejechocard/jep226</pub-id> <pub-id pub-id-type="pmid">20190272</pub-id></citation></ref>
<ref id="B12"><label>12.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Barbier</surname> <given-names>P</given-names></name> <name><surname>Mirea</surname> <given-names>O</given-names></name> <name><surname>Cefalu</surname> <given-names>C</given-names></name> <name><surname>Maltagliati</surname> <given-names>A</given-names></name> <name><surname>Savioli</surname> <given-names>G</given-names></name> <name><surname>Guglielmo</surname> <given-names>M</given-names></name></person-group>. <article-title>Reliability and feasibility of longitudinal AFI global and segmental strain compared with 2D left ventricular volumes and ejection fraction: intra-and inter-operator, test&#x2013;retest, and inter-cycle reproducibility.</article-title> <source><italic>Eur Heart J Cardiovasc Imaging.</italic></source> (<year>2015</year>) <volume>16</volume>:<fpage>642</fpage>&#x2013;<lpage>52</lpage>. <pub-id pub-id-type="doi">10.1093/ehjci/jeu274</pub-id> <pub-id pub-id-type="pmid">25564395</pub-id></citation></ref>
<ref id="B13"><label>13.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Litjens</surname> <given-names>G</given-names></name> <name><surname>Ciompi</surname> <given-names>F</given-names></name> <name><surname>Wolterink</surname> <given-names>J</given-names></name> <name><surname>de Vos</surname> <given-names>B</given-names></name> <name><surname>Leiner</surname> <given-names>T</given-names></name> <name><surname>Teuwen</surname> <given-names>J</given-names></name><etal/></person-group> <article-title>State-of-the-art deep learning in cardiovascular image analysis.</article-title> <source><italic>JACC Cardiovasc Imaging.</italic></source> (<year>2019</year>) <volume>12(8 Pt 1)</volume>:<fpage>1549</fpage>&#x2013;<lpage>65</lpage>. <pub-id pub-id-type="doi">10.1016/j.jcmg.2019.06.009</pub-id> <pub-id pub-id-type="pmid">31395244</pub-id></citation></ref>
<ref id="B14"><label>14.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>J</given-names></name> <name><surname>Gajjala</surname> <given-names>S</given-names></name> <name><surname>Agrawal</surname> <given-names>P</given-names></name> <name><surname>Tison</surname> <given-names>G</given-names></name> <name><surname>Hallock</surname> <given-names>L</given-names></name> <name><surname>Beussink-Nelson</surname> <given-names>L</given-names></name><etal/></person-group> <article-title>Fully automated echocardiogram interpretation in clinical practice: feasibility and diagnostic accuracy.</article-title> <source><italic>Circulation.</italic></source> (<year>2018</year>) <volume>138</volume>:<fpage>1623</fpage>&#x2013;<lpage>35</lpage>. <pub-id pub-id-type="doi">10.1161/CIRCULATIONAHA.118.034338</pub-id> <pub-id pub-id-type="pmid">30354459</pub-id></citation></ref>
<ref id="B15"><label>15.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>C</given-names></name> <name><surname>Xu</surname> <given-names>L</given-names></name> <name><surname>Gao</surname> <given-names>Z</given-names></name> <name><surname>Zhao</surname> <given-names>S</given-names></name> <name><surname>Zhang</surname> <given-names>H</given-names></name> <name><surname>Zhang</surname> <given-names>Y</given-names></name><etal/></person-group> <article-title>Direct delineation of myocardial infarction without contrast agents using a joint motion feature learning architecture.</article-title> <source><italic>Med Image Anal.</italic></source> (<year>2018</year>) <volume>50</volume>:<fpage>82</fpage>&#x2013;<lpage>94</lpage>. <pub-id pub-id-type="doi">10.1016/j.media.2018.09.001</pub-id> <pub-id pub-id-type="pmid">30227385</pub-id></citation></ref>
<ref id="B16"><label>16.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>P</given-names></name> <name><surname>Chen</surname> <given-names>X</given-names></name> <name><surname>Chen</surname> <given-names>E</given-names></name> <name><surname>Yu</surname> <given-names>H</given-names></name> <name><surname>Chen</surname> <given-names>T</given-names></name> <name><surname>Sun</surname> <given-names>S</given-names></name></person-group> <role>editors.</role> <article-title>Anatomy-aware cardiac motion estimation.</article-title> In: <person-group person-group-type="editor"><name><surname>Chen</surname> <given-names>P</given-names></name> <name><surname>Chen</surname> <given-names>X</given-names></name> <name><surname>Chen</surname> <given-names>E</given-names></name> <name><surname>Yu</surname> <given-names>H</given-names></name> <name><surname>Chen</surname> <given-names>T</given-names></name> <name><surname>Sun</surname> <given-names>S</given-names></name></person-group> <role>editors.</role> <source><italic>International Workshop on Machine Learning in Medical Imaging.</italic></source> <publisher-loc>Berlin</publisher-loc>: <publisher-name>Springer</publisher-name> (<year>2020</year>). <pub-id pub-id-type="doi">10.1007/978-3-030-59861-7_16</pub-id></citation></ref>
<ref id="B17"><label>17.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vos</surname> <given-names>B</given-names></name> <name><surname>Berendsen</surname> <given-names>F</given-names></name> <name><surname>Viergever</surname> <given-names>M</given-names></name> <name><surname>Staring</surname> <given-names>M</given-names></name> <name><surname>I&#x0161;gum</surname> <given-names>I.</given-names></name></person-group> <source><italic>End-to-End Unsupervised deformable Image Registration With a Convolutional Neural Network. Deep Learning In Medical Image Analysis and Multimodal Learning for Clinical Decision Support.</italic></source> <publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name> (<year>2017</year>). <fpage>p. 204</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-319-67558-9_24</pub-id></citation></ref>
<ref id="B18"><label>18.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>H</given-names></name> <name><surname>Sun</surname> <given-names>S</given-names></name> <name><surname>Yu</surname> <given-names>H</given-names></name> <name><surname>Chen</surname> <given-names>X</given-names></name> <name><surname>Shi</surname> <given-names>H</given-names></name> <name><surname>Huang</surname> <given-names>T</given-names></name><etal/></person-group> <role>editors.</role> <article-title>Foal: fast online adaptive learning for cardiac motion estimation.</article-title> <source><italic>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition.</italic></source> <publisher-loc>Seattle, WA</publisher-loc>: <publisher-name>IEEE</publisher-name> (<year>2020</year>). <fpage>p. 4312</fpage>&#x2013;<lpage>22</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR42600.2020.00437</pub-id></citation></ref>
<ref id="B19"><label>19.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>&#x00D8;stvik</surname> <given-names>A</given-names></name> <name><surname>Salte</surname> <given-names>I</given-names></name> <name><surname>Smistad</surname> <given-names>E</given-names></name> <name><surname>Nguyen</surname> <given-names>T</given-names></name> <name><surname>Melichova</surname> <given-names>D</given-names></name> <name><surname>Brunvand</surname> <given-names>H</given-names></name><etal/></person-group> <article-title>Myocardial function imaging in echocardiography using deep learning.</article-title> <source><italic>IEEE Trans Med Imaging.</italic></source> (<year>2021</year>) <volume>40</volume>:<fpage>1340</fpage>&#x2013;<lpage>51</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2021.3054566</pub-id> <pub-id pub-id-type="pmid">33493114</pub-id></citation></ref>
<ref id="B20"><label>20.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>D</given-names></name> <name><surname>Yang</surname> <given-names>X</given-names></name> <name><surname>Liu</surname> <given-names>M</given-names></name> <name><surname>Kautz</surname> <given-names>J</given-names></name></person-group> <role>editors.</role> <article-title>Pwc-net: Cnns for optical flow using pyramid, warping, and cost volume.</article-title> <source><italic>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition.</italic></source> <publisher-loc>Salt Lake City, UT</publisher-loc>: (<year>2018</year>). <pub-id pub-id-type="doi">10.1109/CVPR.2018.00931</pub-id></citation></ref>
<ref id="B21"><label>21.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Salte</surname> <given-names>I</given-names></name> <name><surname>&#x00D8;stvik</surname> <given-names>A</given-names></name> <name><surname>Smistad</surname> <given-names>E</given-names></name> <name><surname>Melichova</surname> <given-names>D</given-names></name> <name><surname>Nguyen</surname> <given-names>T</given-names></name> <name><surname>Karlsen</surname> <given-names>S</given-names></name><etal/></person-group> <article-title>Artificial intelligence for automatic measurement of left ventricular strain in echocardiography.</article-title> <source><italic>Cardiovasc Imaging.</italic></source> (<year>2021</year>) <volume>14</volume>:<fpage>1918</fpage>&#x2013;<lpage>28</lpage>. <pub-id pub-id-type="doi">10.1016/j.jcmg.2021.04.018</pub-id> <pub-id pub-id-type="pmid">34147442</pub-id></citation></ref>
<ref id="B22"><label>22.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ouyang</surname> <given-names>D</given-names></name> <name><surname>He</surname> <given-names>B</given-names></name> <name><surname>Ghorbani</surname> <given-names>A</given-names></name> <name><surname>Yuan</surname> <given-names>N</given-names></name> <name><surname>Ebinger</surname> <given-names>J</given-names></name> <name><surname>Langlotz</surname> <given-names>C</given-names></name><etal/></person-group> <article-title>Video-based AI for beat-to-beat assessment of cardiac function.</article-title> <source><italic>Nature.</italic></source> (<year>2020</year>) <volume>580</volume>:<fpage>252</fpage>&#x2013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-020-2145-8</pub-id> <pub-id pub-id-type="pmid">32269341</pub-id></citation></ref>
<ref id="B23"><label>23.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>C</given-names></name> <name><surname>Qin</surname> <given-names>C</given-names></name> <name><surname>Qiu</surname> <given-names>H</given-names></name> <name><surname>Tarroni</surname> <given-names>G</given-names></name> <name><surname>Duan</surname> <given-names>J</given-names></name> <name><surname>Bai</surname> <given-names>W</given-names></name><etal/></person-group> <article-title>Deep learning for cardiac image segmentation: a review.</article-title> <source><italic>Front Cardiovasc Med.</italic></source> (<year>2020</year>) <volume>7</volume>:<issue>25</issue>. <pub-id pub-id-type="doi">10.3389/fcvm.2020.00025</pub-id> <pub-id pub-id-type="pmid">32195270</pub-id></citation></ref>
<ref id="B24"><label>24.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Teed</surname> <given-names>Z</given-names></name> <name><surname>Deng</surname> <given-names>J editors</given-names></name></person-group>. <article-title>Raft: recurrent all-pairs field transforms for optical flow.</article-title> In: <person-group person-group-type="editor"><name><surname>Vedaldi</surname> <given-names>A</given-names></name> <name><surname>Bischof</surname> <given-names>H</given-names></name> <name><surname>Brox</surname> <given-names>T</given-names></name> <name><surname>Frahm</surname> <given-names>J-M</given-names></name></person-group> <role>editors.</role> <source><italic>European Conference on Computer Vision.</italic></source> <publisher-loc>Berlin</publisher-loc>: <publisher-name>Springer</publisher-name> (<year>2020</year>). <pub-id pub-id-type="doi">10.24963/ijcai.2021/662</pub-id></citation></ref>
<ref id="B25"><label>25.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>&#x00D8;stvik</surname> <given-names>A</given-names></name> <name><surname>Smistad</surname> <given-names>E</given-names></name> <name><surname>Espeland</surname> <given-names>T</given-names></name> <name><surname>Berg</surname> <given-names>E</given-names></name> <name><surname>Lovstakken</surname> <given-names>L</given-names></name> <name><surname>Stoyanov</surname> <given-names>D</given-names></name><etal/></person-group> <role>editors.</role> <article-title>Automatic myocardial strain imaging in echocardiography using deep learning.</article-title> <source><italic>Deep Learning in Medical Image Analysis and Multimodal Learning for Clinical Decision Support.</italic></source> <publisher-loc>Berlin</publisher-loc>: <publisher-name>Springer</publisher-name> (<year>2018</year>). <fpage>p. 309</fpage>&#x2013;<lpage>16</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-030-00889-5_35</pub-id></citation></ref>
<ref id="B26"><label>26.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Diba</surname> <given-names>A</given-names></name> <name><surname>Fayyaz</surname> <given-names>M</given-names></name> <name><surname>Sharma</surname> <given-names>V</given-names></name> <name><surname>Karami</surname> <given-names>A</given-names></name> <name><surname>Arzani</surname> <given-names>M</given-names></name> <name><surname>Yousefzadeh</surname> <given-names>R</given-names></name><etal/></person-group> <article-title>Temporal 3d convnets: new architecture and transfer learning for video classification.</article-title> <source><italic>arXiv</italic></source> [<comment>Preprint</comment>]. (<year>2017</year>). <pub-id pub-id-type="doi">10.48550/arXiv.1711.08200</pub-id></citation></ref>
<ref id="B27"><label>27.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dey</surname> <given-names>R</given-names></name> <name><surname>Salem</surname> <given-names>F</given-names></name></person-group> <role>editors.</role> <article-title>Gate-variants of gated recurrent unit (GRU) neural networks.</article-title> <source><italic>Proceedings of the 2017 IEEE 60th International Midwest Symposium on Circuits and Systems (MWSCAS).</italic></source> <publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name> (<year>2017</year>). <pub-id pub-id-type="doi">10.1109/MWSCAS.2017.8053243</pub-id></citation></ref>
<ref id="B28"><label>28.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Voigt</surname> <given-names>J</given-names></name> <name><surname>Pedrizzetti</surname> <given-names>G</given-names></name> <name><surname>Lysyansky</surname> <given-names>P</given-names></name> <name><surname>Marwick</surname> <given-names>T</given-names></name> <name><surname>Houle</surname> <given-names>H</given-names></name> <name><surname>Baumann</surname> <given-names>R</given-names></name><etal/></person-group> <article-title>Definitions for a common standard for 2D speckle tracking echocardiography: consensus document of the EACVI/ASE/Industry task force to standardize deformation imaging.</article-title> <source><italic>Eur Heart J Cardiovasc Imaging.</italic></source> (<year>2015</year>) <volume>16</volume>:<fpage>1</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1093/ehjci/jeu184</pub-id> <pub-id pub-id-type="pmid">25525063</pub-id></citation></ref>
<ref id="B29"><label>29.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alessandrini</surname> <given-names>M</given-names></name> <name><surname>Chakraborty</surname> <given-names>B</given-names></name> <name><surname>Heyde</surname> <given-names>B</given-names></name> <name><surname>Bernard</surname> <given-names>O</given-names></name> <name><surname>De Craene</surname> <given-names>M</given-names></name> <name><surname>Sermesant</surname> <given-names>M</given-names></name><etal/></person-group> <article-title>Realistic vendor-specific synthetic ultrasound data for quality assurance of 2-D speckle tracking echocardiography: simulation pipeline and open access database.</article-title> <source><italic>IEEE Trans Ultrason Ferroelectr Frequency Control.</italic></source> (<year>2017</year>) <volume>65</volume>:<fpage>411</fpage>&#x2013;<lpage>22</lpage>. <pub-id pub-id-type="doi">10.1109/TUFFC.2017.2786300</pub-id> <pub-id pub-id-type="pmid">29505408</pub-id></citation></ref>
<ref id="B30"><label>30.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ghosh</surname> <given-names>S</given-names></name> <name><surname>Das</surname> <given-names>N</given-names></name> <name><surname>Das</surname> <given-names>I</given-names></name> <name><surname>Maulik</surname> <given-names>U</given-names></name></person-group>. <article-title>Understanding deep learning techniques for image segmentation.</article-title> <source><italic>ACM Comput Surv (CSUR).</italic></source> (<year>2019</year>) <volume>52</volume>:<fpage>1</fpage>&#x2013;<lpage>35</lpage>. <pub-id pub-id-type="doi">10.1145/3329784</pub-id></citation></ref>
<ref id="B31"><label>31.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Smistad</surname> <given-names>E</given-names></name> <name><surname>Salte</surname> <given-names>I</given-names></name> <name><surname>&#x00D8;stvik</surname> <given-names>A</given-names></name> <name><surname>Leclerc</surname> <given-names>S</given-names></name> <name><surname>Bernard</surname> <given-names>O</given-names></name> <name><surname>Lovstakken</surname> <given-names>L</given-names></name></person-group> <role>editors.</role> <article-title>Segmentation of apical long axis, four-and two-chamber views using deep neural networks.</article-title> <source><italic>Proceedings of the 2019 IEEE International Ultrasonics Symposium (IUS).</italic></source> <publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name> (<year>2019</year>). <pub-id pub-id-type="doi">10.1109/ULTSYM.2019.8926017</pub-id></citation></ref>
<ref id="B32"><label>32.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Leclerc</surname> <given-names>S</given-names></name> <name><surname>Smistad</surname> <given-names>E</given-names></name> <name><surname>Pedrosa</surname> <given-names>J</given-names></name> <name><surname>&#x00D8;stvik</surname> <given-names>A</given-names></name> <name><surname>Cervenansky</surname> <given-names>F</given-names></name> <name><surname>Espinosa</surname> <given-names>F</given-names></name><etal/></person-group> <article-title>Deep learning for segmentation using an open large-scale dataset in 2D echocardiography.</article-title> <source><italic>IEEE Trans Med Imaging.</italic></source> (<year>2019</year>) <volume>38</volume>:<fpage>2198</fpage>&#x2013;<lpage>210</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2019.2900516</pub-id> <pub-id pub-id-type="pmid">30802851</pub-id></citation></ref>
<ref id="B33"><label>33.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tyomkin</surname> <given-names>V</given-names></name> <name><surname>Vered</surname> <given-names>Z</given-names></name> <name><surname>Mor</surname> <given-names>M</given-names></name> <name><surname>Blondheim</surname> <given-names>D</given-names></name> <name><surname>Carasso</surname> <given-names>S</given-names></name> <name><surname>Shimoni</surname> <given-names>S</given-names></name><etal/></person-group> <article-title>Is it time to revise the guidelines and recommendations for digital echocardiography?</article-title> <source><italic>J Am Soc Echocardiogr.</italic></source> (<year>2018</year>) <volume>31</volume>:<fpage>634</fpage>&#x2013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1016/j.echo.2018.01.021</pub-id> <pub-id pub-id-type="pmid">29573930</pub-id></citation></ref>
<ref id="B34"><label>34.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Khaishagi</surname> <given-names>M</given-names></name> <name><surname>Kumar</surname> <given-names>P</given-names></name> <name><surname>Naik</surname> <given-names>D</given-names></name></person-group> <role>editors.</role> <article-title>Dense optical flow using RAFT.</article-title> <source><italic>Proceedings of the 2022 IEEE Fourth International Conference on Advances in Electronics, Computers and Communications (ICAECC).</italic></source> <publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name> (<year>2022</year>). <pub-id pub-id-type="doi">10.1109/ICAECC54045.2022.9716703</pub-id> <pub-id pub-id-type="pmid">36286373</pub-id></citation></ref>
<ref id="B35"><label>35.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>D</given-names></name> <name><surname>Herrmann</surname> <given-names>C</given-names></name> <name><surname>Reda</surname> <given-names>F</given-names></name> <name><surname>Rubinstein</surname> <given-names>M</given-names></name> <name><surname>Fleet</surname> <given-names>D</given-names></name> <name><surname>Freeman</surname> <given-names>W</given-names></name></person-group>. <article-title>What makes RAFT better than PWC-net?</article-title> <source><italic>arXiv</italic></source> [<comment>Preprint</comment>]. (<year>2022</year>). <volume>arXiv</volume>:<issue>220310712</issue>.</citation></ref>
<ref id="B36"><label>36.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shakir</surname> <given-names>D</given-names></name> <name><surname>Rasul</surname> <given-names>K</given-names></name></person-group>. <article-title>Chemotherapy induced cardiomyopathy: pathogenesis, monitoring and management.</article-title> <source><italic>J Clin Med Res.</italic></source> (<year>2009</year>) <volume>1</volume>:<fpage>8</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.4021/jocmr2009.02.1225</pub-id> <pub-id pub-id-type="pmid">22505958</pub-id></citation></ref>
<ref id="B37"><label>37.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dellinger</surname> <given-names>R</given-names></name> <name><surname>Levy</surname> <given-names>M</given-names></name> <name><surname>Rhodes</surname> <given-names>A</given-names></name> <name><surname>Annane</surname> <given-names>D</given-names></name> <name><surname>Gerlach</surname> <given-names>H</given-names></name> <name><surname>Opal</surname> <given-names>S</given-names></name><etal/></person-group> <article-title>Surviving sepsis campaign: international guidelines for management of severe sepsis and septic shock, 2012.</article-title> <source><italic>Intensive Care Med.</italic></source> (<year>2013</year>) <volume>39</volume>:<fpage>165</fpage>&#x2013;<lpage>228</lpage>. <pub-id pub-id-type="doi">10.1007/s00134-012-2769-8</pub-id> <pub-id pub-id-type="pmid">23361625</pub-id></citation></ref>
<ref id="B38"><label>38.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Asch</surname> <given-names>F</given-names></name> <name><surname>Poilvert</surname> <given-names>N</given-names></name> <name><surname>Abraham</surname> <given-names>T</given-names></name> <name><surname>Jankowski</surname> <given-names>M</given-names></name> <name><surname>Cleve</surname> <given-names>J</given-names></name> <name><surname>Adams</surname> <given-names>M</given-names></name><etal/></person-group> <article-title>Automated echocardiographic quantification of left ventricular ejection fraction without volume measurements using a machine learning algorithm mimicking a human expert.</article-title> <source><italic>Circ Cardiovasc Imaging.</italic></source> (<year>2019</year>) <volume>12</volume>:<issue>e009303</issue>. <pub-id pub-id-type="doi">10.1161/CIRCIMAGING.119.009303</pub-id> <pub-id pub-id-type="pmid">31522550</pub-id></citation></ref>
<ref id="B39"><label>39.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ghorbani</surname> <given-names>A</given-names></name> <name><surname>Ouyang</surname> <given-names>D</given-names></name> <name><surname>Abid</surname> <given-names>A</given-names></name> <name><surname>He</surname> <given-names>B</given-names></name> <name><surname>Chen</surname> <given-names>J</given-names></name> <name><surname>Harrington</surname> <given-names>R</given-names></name><etal/></person-group> <article-title>Deep learning interpretation of echocardiograms.</article-title> <source><italic>NPJ Digit Med.</italic></source> (<year>2020</year>) <volume>3</volume>:<issue>10</issue>. <pub-id pub-id-type="doi">10.1038/s41746-019-0216-8</pub-id> <pub-id pub-id-type="pmid">31993508</pub-id></citation></ref>
</ref-list>
</back>
</article>
