<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurosci.</journal-id>
<journal-title>Frontiers in Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurosci.</abbrev-journal-title>
<issn pub-type="epub">1662-453X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnins.2022.1117134</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Reliable and stable fundus image registration based on brain-inspired spatially-varying adaptive pyramid context aggregation network</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Xu</surname> <given-names>Jie</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2127081/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Yang</surname> <given-names>Kang</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2133178/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Chen</surname> <given-names>Youxin</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1542270/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Dai</surname> <given-names>Liming</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhang</surname> <given-names>Dongdong</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Shuai</surname> <given-names>Ping</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1934771/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Shi</surname> <given-names>Rongjie</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Yang</surname> <given-names>Zhanbo</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Beijing Institute of Ophthalmology, Beijing Tongren Eye Center, Beijing Tongren Hospital, Capital Medical University, Beijing Key Laboratory of Ophthalmology and Visual Sciences</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Beijing Zhizhen Internet Technology Co. Ltd.,</institution> <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Department of Ophthalmology, Peking Union Medical College Hospital</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>Department of Health Management and Physical Examination, Sichuan Provincial People&#x00027;s Hospital, University of Electronic Science and Technology of China</institution>, <addr-line>Chengdu</addr-line>, <country>China</country></aff>
<aff id="aff5"><sup>5</sup><institution>School of Medicine, University of Electronic Science and Technology of China</institution>, <addr-line>Chengdu</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Chenwei Deng, Beijing Institute of Technology, China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Jian Jia, University of Chinese Academy of Sciences, China; Anca Marginean, Technical University of Cluj-Napoca, Romania; Wenzhen Haung, Tsinghua University, China; Qiaozhe Li, Chinese Academy of Sciences (CAS), China</p></fn>

<corresp id="c001">&#x0002A;Correspondence: Youxin Chen &#x02709; <email>chenyx&#x00040;pumch.cn</email></corresp>
<fn fn-type="other" id="fn001"><p>This article was submitted to Perception Science, a section of the journal Frontiers in Neuroscience</p></fn></author-notes>
<pub-date pub-type="epub">
<day>16</day>
<month>01</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>16</volume>
<elocation-id>1117134</elocation-id>
<history>
<date date-type="received">
<day>06</day>
<month>12</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>28</day>
<month>12</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2023 Xu, Yang, Chen, Dai, Zhang, Shuai, Shi and Yang.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Xu, Yang, Chen, Dai, Zhang, Shuai, Shi and Yang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license> </permissions>
<abstract>
<p>The task of fundus image registration aims to find matching keypoints between an image pair. Traditional methods detect the keypoint by hand-designed features, which fail to cope with complex application scenarios. Due to the strong feature learning ability of deep neural network, current image registration methods based on deep learning directly learn to align the geometric transformation between the reference image and test image in an end-to-end manner. Another mainstream of this task aims to learn the displacement vector field between the image pair. In this way, the image registration has achieved significant advances. However, due to the complicated vascular morphology of retinal image, such as texture and shape, current widely used image registration methods based on deep learning fail to achieve reliable and stable keypoint detection and registration results. To this end, in this paper, we aim to bridge this gap. Concretely, since the vessel crossing and branching points can reliably and stably characterize the key components of fundus image, we propose to learn to detect and match all the crossing and branching points of the input images based on a single deep neural network. Moreover, in order to accurately locate the keypoints and learn discriminative feature embedding, a brain-inspired spatially-varying adaptive pyramid context aggregation network is proposed to incorporate the contextual cues under the supervision of structured triplet ranking loss. Experimental results show that the proposed method achieves more accurate registration results with significant speed advantage.</p></abstract>
<kwd-group>
<kwd>retinal image analysis</kwd>
<kwd>fundus image registration</kwd>
<kwd>deep learning</kwd>
<kwd>context aggregation</kwd>
<kwd>structured triplet ranking loss</kwd>
</kwd-group>
<counts>
<fig-count count="8"/>
<table-count count="4"/>
<equation-count count="9"/>
<ref-count count="48"/>
<page-count count="13"/>
<word-count count="8822"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1. Introduction</title>
<p>Fundus image analysis has been widely researched, due to its significant advantage of non-invasive observation. The purpose of image registration (Hill et al., <xref ref-type="bibr" rid="B14">2001</xref>; Sotiras et al., <xref ref-type="bibr" rid="B38">2013</xref>) is to deform the test image to the coordinate system of the reference image, so that the same point can be imaged at the same coordinate of the two images (Oliveira and Tavares, <xref ref-type="bibr" rid="B27">2014</xref>). Registration of medical images is a crucial step in the image processing. Image registration can trace the progression of the same patient through time, providing a basis for clinical diagnosis, lowering physician effort, and aiding in the investigation of disease prognosis and outcome. In order to accurately learn the deformation coefficient to transform the test image, the matching keypoints between the test image and reference image should be obtained. To this end, previous methods rely on human-designed features to distinguish among visually similar keypoints, by encoding the texture, shape or intensity gradient with particularly designed computing pattern. Recently, deep neural network (DNN) (Krizhevsky et al., <xref ref-type="bibr" rid="B18">2012</xref>; Simonyan and Zisserman, <xref ref-type="bibr" rid="B35">2014</xref>; He et al., <xref ref-type="bibr" rid="B10">2016</xref>) based image registration has made rapid progress due to its strong feature learning ability. Some current DNN based methods propose to directly learn the geometric transformation, such as homography transformation, between the test image and reference image. Other works also aim to learn the dense pixel-level displacement vector filed between the image pair (Cao et al., <xref ref-type="bibr" rid="B3">2017</xref>; Krebs et al., <xref ref-type="bibr" rid="B17">2017</xref>). However, due to the complex and variable retinal vascular structure, previous methods fail to achieve <bold>reliable</bold> and <bold>stable</bold> registration performance, which severely limits downstream applications. Considering that the vessel crossing and branching points are able to reliably and stably characterize the fundus image (Deng et al., <xref ref-type="bibr" rid="B7">2010</xref>; Chen et al., <xref ref-type="bibr" rid="B4">2011</xref>), we propose to choose all the crossing and branching points as the keypoints. To this end, a single deep neural network is utilized to learn to simultaneously locate and match all the keypoints.</p>
<p>Since the lower-level spatial details and higher-level semantic cues of fundus image are both critical for learning accurate keypoint detection and corresponding discriminative feature embedding for keypoint matching, we employ the widely used encoder-decoder architecture (Ronneberger et al., <xref ref-type="bibr" rid="B32">2015</xref>) as the basic network. Moreover, due to the large intra-class variability and small inter-class difference of fundus image, the non-matching keypoints are prone to be misclassified. It is natural for human being to gain the knowledge of contextual consistency, which is helpful for alleviating this issue. As a result, contextual cues should be incorporated into the vanilla encoder-decoder architecture to handle these critical issues. To this end, on the basis of the encoder-decoder architecture, we propose a brain-inspired spatially-varying adaptive pyramid context aggregation network. Concretely, with the proposed spatially-varying adaptive pyramid context aggregation module, every pixel location of the feature map is reweighted with the learned weight factor guided by the aggregated global contextual cues. Feature vectors of any two pixel locations are explicitly interacted by the form of matrix multiplication between the reshaped two-dimensional feature maps, leading to the spatially-varying feature weight factors. The generated weight factors are then utilized as the dilated depth-wise convolution kernels with different dilation factors to aggregate the contextual cues in receptive fields with multiple scales. In this way, the contextual cues are integrated into the feature maps with predictable and spatially-varying depth-wise convolutions. In addition, we employ a structured triplet ranking loss, whose aim is to supervise the network to enlarge the distance of feature embedding between non-matching keypoints and narrow the distance of feature embedding between the matching keypoints, leading to compactness between matching keypoints and dispersion between non-matching keypoints.</p>
<p>In order to verify the effectiveness of the proposed method, proper dataset and evaluation metric should be elaborately designed. However, current FIRE dataset (Hernandez-Matas et al., <xref ref-type="bibr" rid="B12">2017</xref>) only labels a small part of the keypoints. Meanwhile, some keypoints of FIRE dataset are not located at branching or crossing points. So this dataset can&#x00027;t be used for training our proposed model. To this end, we collect 200 retinal images of 50 patients taken with fundus camera by RetCam3 and Canon. Concretely, 100 neonatal fundus images of 27 patients with low imaging quality are taken from RetCam3. Another 100 high-quality retinal images of 23 patients taken from Canon are also included. Meanwhile, different imaging angles and diverse overlapping areas between the image pair are also considered during the construction of dataset. In order to quantitatively evaluate the proposed method, following previous methods (Hernandez-Matas et al., <xref ref-type="bibr" rid="B12">2017</xref>), we choose the Area Under Curve (AUC) value as the registration score. Experimental results demonstrate that our proposed method achieves significant performance improvement over the vanilla encoder-decoder network. Our method achieves the best registration performance among the deep learning based methods. Meanwhile, our proposed method also surpasses most of the traditional registration methods with significantly faster execution speed by an order of magnitude.</p>
<p>Our contributions are summarized into three parts:</p>
<list list-type="bullet">
<list-item><p>We propose to achieve <bold>reliable</bold> and <bold>stable</bold> keypoint detection and registration results for fundus image. Considering that the vessel crossing and branching points can reliably and stably characterize the key components of fundus image, we propose to learn to detect and match all the crossing and branching points of the input image pair with a single deep neural network.</p></list-item>
<list-item><p>In order to cope with the large intra-class variability and small inter-class difference of retinal image, we propose a brain-inspired spatially-varying adaptive pyramid context aggregation based on the widely used encoder-decoder architecture. In this way, long-range contextual cues are incorporated into the feature maps with predictable and input-variant convolutions. Moreover, a structured triplet ranking loss is employed to enforce the network to produce similar feature embedding for matching keypoints in the input image pair, and dissimilar feature embedding for non-matching keypoints.</p></list-item>
<list-item><p>Since there is no proper fundus image registration dataset for method evaluation, we construct a large-scale dataset which covers diverse application scenarios. Quantitative and qualitative results show that our proposed method is able to reliably and stably locate and match keypoints.</p></list-item>
</list>
<p>We organize our paper as follows. Section 2 reviews related work. Section 3 shows the detail of our method. Section 4 demonstrates experimental results. Finally, Section 5 presents our conclusion.</p>
</sec>
<sec id="s2">
<title>2. Related work</title>
<sec>
<title>2.1. Deep learning based image registration</title>
<p>Since the learning based image registration is mainly considered in this paper, we provide a brief review of related works on deep learning based image registration in this part. In recent years, several methods (Cao et al., <xref ref-type="bibr" rid="B3">2017</xref>; Krebs et al., <xref ref-type="bibr" rid="B17">2017</xref>) have proposed to employ the DNN to directly learn the warp field between the test image and reference image. Ground truth warp fields are required in the above methods (Roh&#x000E9; et al., <xref ref-type="bibr" rid="B31">2017</xref>; Sokooti et al., <xref ref-type="bibr" rid="B37">2017</xref>; Yang et al., <xref ref-type="bibr" rid="B46">2017</xref>) to supervise the learning of DNN. In order to obtain the ground truth warp field, several methods propose to simulate the deformation operation and generate deformed images. Some other methods employ the classical registration method, which rely on hand-designed feature. However, the above methods are difficult to obtain ground truth warp field as the ground reality, which severely limit the application in real scenario. Recently, several unsupervised learning based image registration methods (Li and Fan, <xref ref-type="bibr" rid="B20">2017</xref>; Vos et al., <xref ref-type="bibr" rid="B40">2017</xref>; Zou et al., <xref ref-type="bibr" rid="B48">2020</xref>) are also proposed. However, these methods fail to cope with complex image registration application, such as large transformations (Vos et al., <xref ref-type="bibr" rid="B40">2017</xref>).</p>
<p>Compared to images collected in our daily life, the retinal image registration is a much more challenging problem. First, there are large differences in illumination, color, contrast and imaging angles of the input image pair in diverse scenarios. The overlapping areas between the test image and the reference image may be also diverse. Furthermore, significant changes in retinal structure may be caused by the progression of retinopathy. As a result, current deep learning based image registration methods fail to achieve reliable and stable registration results, which are not applicable for the challenging fundus image task.</p>
</sec>
<sec>
<title>2.2. Deep metric learning</title>
<p>Deep metric learning aims to learn the distance metric to compare and measure similarity between pairs of examples, which is important for various tasks, such as image retrieval (Sohn, <xref ref-type="bibr" rid="B36">2016</xref>; Movshovitz-Attias et al., <xref ref-type="bibr" rid="B24">2017</xref>), clustering (Hershey et al., <xref ref-type="bibr" rid="B13">2016</xref>). One of the main task of deep metric learning is to design proper loss function. Contrastive loss (Chopra et al., <xref ref-type="bibr" rid="B5">2005</xref>; Hadsell et al., <xref ref-type="bibr" rid="B9">2006</xref>) aims to encode the pair-wise relations between the anchor example and one similar(positive) or dissimilar(negative) example, which is first proposed to learn the feature embedding for image search task. Triplet loss (Wang et al., <xref ref-type="bibr" rid="B42">2014</xref>; Schroff et al., <xref ref-type="bibr" rid="B33">2015</xref>; Cui et al., <xref ref-type="bibr" rid="B6">2016</xref>) is used to learn feature embedding for face recognition task. A triplet is composed of the anchor example, a positive example and a negative example. The triplet loss is to learn a distance metric by which the anchor point is closer to the similar point than the dissimilar one by a margin. Recently, richer structural relations among multiple examples are considered by ranking-motivated methods (Schroff et al., <xref ref-type="bibr" rid="B33">2015</xref>; Oh Song et al., <xref ref-type="bibr" rid="B26">2016</xref>; Sohn, <xref ref-type="bibr" rid="B36">2016</xref>; Law et al., <xref ref-type="bibr" rid="B19">2017</xref>; Movshovitz-Attias et al., <xref ref-type="bibr" rid="B24">2017</xref>). Some other methods propose to design clustering-motivated structured losses (Hershey et al., <xref ref-type="bibr" rid="B13">2016</xref>; Oh Song et al., <xref ref-type="bibr" rid="B25">2017</xref>). However, since clustering-motivated losses are more difficult to optimize, the ranking-motivated loss function is mainly considered in this paper.</p>
</sec>
</sec>
<sec id="s3">
<title>3. Method details</title>
<p>This section presents details of our method for reliable and stable fundus image registration. We show the overview of the proposed model in <xref ref-type="fig" rid="F1">Figure 1</xref>. We start by introducing the encoder-decoder network, which is the baseline of our model. Then we introduce the proposed network architecture and employed loss function.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Overview of the proposed network for simultaneous keypoints detection and keypoints matching.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-16-1117134-g0001.tif"/>
</fig>
<sec>
<title>3.1. Encoder-decoder network architecture</title>
<p>For the fundus image registration method based on deep neural network (DNN), in order to achieve accurate pixel-level image registration results, robust global semantic information and rich local spatial details are required. Current DNN stacks successive convolutional and pooling layers to obtain roust feature representations. However, due to the multiple pooling operations, the feature spatial resolution is largely reduced. As a result, local spatial details are severely lost for the features in deeper-level layers. On the contrary, due to fewer pooling layers, spatial resolution of features in lower-level layers are larger. In this way, the features in lower-level layers encode rich local spatial details. However, the lack of semantic and discriminative cues make the lower-level features fail to effectively model long-range information. Since both the local spatial details and global semantic cues are essential for accurate image registration performance, a balanced fusion of the lower-level features and the deeper-level features is required.</p>
<p>As shown in <xref ref-type="fig" rid="F1">Figure 1</xref>, current widely used encoder-decoder architecture employs the encoder sub-network to extract the multi-scale features by multiple stacked convolutional and pooling operations. The later decoder sub-network then combines the extracted multi-level features by multiple feature fusion operations. Concretely, with the input image pair, the successive convolutional and pooling layers of encoder sub-network extract multi-scale features, similar to ResNet (He et al., <xref ref-type="bibr" rid="B10">2016</xref>) or VGGNet (Simonyan and Zisserman, <xref ref-type="bibr" rid="B35">2014</xref>). The decoder sub-network consists of multiple feature fusion operations, which are employed to fuse the multi-scale features generated by the encoder sub-network progressively. For every fusion operation, <inline-formula><mml:math id="M1"><mml:mover accent="true"><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:math></inline-formula>, the feature in current layer <italic>i</italic>, is first upsampled to match the resolution of the feature map <italic>F</italic><sub><italic>i</italic>&#x02212;1</sub> from the lower neighbor layer <italic>i</italic>&#x02212;1. The feature concatenation along the channel dimension is applied, which is followed by another convolution for further feature abstraction. This operation can be formulated as:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mrow><mml:msubsup><mml:mi>F</mml:mi><mml:mrow><mml:msub><mml:mrow><mml:mtext>&#x000A0;</mml:mtext></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mtext>&#x005E;</mml:mtext></mml:msubsup><mml:mo>=</mml:mo><mml:mi>C</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi>v</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>C</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>U</mml:mi><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mover accent='true'><mml:mi>F</mml:mi><mml:mo>&#x005E;</mml:mo></mml:mover><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mo>.</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The above fusion operation is iterated until the lowest layer, where the generated feature <italic>F</italic><sub>1</sub> has the same spatial resolution as the input image, which is used to produce the final prediction.</p>
</sec>
<sec>
<title>3.2. Spatially-varying context aggregation module</title>
<p>Due to the large intra-class variability and small inter-class difference of fundus image, the non-matching keypoints are prone to be misclassified. As a result, contextual cues should be incorporated into the vanilla encoder-decoder architecture to handle this critical issue (Liu et al., <xref ref-type="bibr" rid="B21">2020</xref>). To this end, with the deepest feature map generated by the encoder, a novel context aggregation module is applied to incorporate the contextual cues in a spatially-varying manner. The details are illustrated below.</p>
<p>In order to model the long-range contextual cues, previous methods are mainly designed to generate global-consistent feature re-weighting coefficient. For example, SE-Net (Jie et al., <xref ref-type="bibr" rid="B16">2019</xref>) is proposed to produce channel-wise feature re-weighting factor of global distribution by a squeeze-and-excitation mechanism. Differently, we propose to aggregate the global contextual cues by generating spatially-varying feature re-weighting factors. In this way, the long-range relations are more effectively mined in a spatially-varying manner.</p>
<p><xref ref-type="fig" rid="F2">Figure 2</xref> shows the overall architecture of the proposed Spatially-varying Context Aggregation (SCA) module. First, we explicitly model the long-range relations between any two pixel locations by matrix multiplication, generating spatially-varying context kernel prediction. Then, the predicted context kernels are applied on the original feature map, leading to aggregated context enhanced feature. Following are the detailed processing pipeline.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Details of the proposed spatially-varying context aggregation module, which first predicts the spatially-varying context kernel and then aggregates the context with the predicted re-weight kernels.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-16-1117134-g0002.tif"/>
</fig>
<p>With the feature map <italic>X</italic> &#x02208; <italic>R</italic><sup><italic>H</italic>&#x000D7;<italic>W</italic>&#x000D7;<italic>C</italic></sup> generated by the last feature block of the encoder, we first transform it into two forms with two independent convolutional operations: the <italic>key</italic> and <italic>query</italic>. The <italic>H</italic>, <italic>W</italic> and <italic>C</italic> refer to the hight, width and channel number, respectively. The <italic>key</italic> feature map <italic>K</italic> &#x02208; <italic>R</italic><sup><italic>H</italic>&#x000D7;<italic>W</italic>&#x000D7;<italic>C</italic></sup> and the <italic>query</italic> feature map <italic>Q</italic> &#x02208; <italic>R</italic><sup><italic>H</italic>&#x000D7;<italic>W</italic>&#x000D7;<italic>s</italic><sup>2</sup></sup> are then used to aggregate the contextual cues. Here, <italic>s</italic> is the kernel size of the learned context kernel.</p>
<p>In order to effectively model the global contextual cues between pixels, the relation within any pixel locations should be explicitly interacted. To this end, the <italic>key</italic> feature map <italic>K</italic> &#x02208; <italic>R</italic><sup><italic>H</italic>&#x000D7;<italic>W</italic>&#x000D7;<italic>C</italic></sup> and the <italic>query</italic> feature map <italic>Q</italic> &#x02208; <italic>R</italic><sup><italic>H</italic>&#x000D7;<italic>W</italic>&#x000D7;<italic>s</italic><sup>2</sup></sup> are first reshaped into 2D form, <italic>K</italic> &#x02208; <italic>Q</italic><sup><italic>H</italic>&#x000D7;<italic>W</italic>&#x000D7;<italic>C</italic></sup> and <italic>Q</italic>&#x02032; &#x02208; <italic>R</italic><sup>(<italic>H</italic>&#x000D7;<italic>W</italic>) &#x000D7; <italic>s</italic><sup>2</sup></sup>, respectively. In this way, our aim is to make each column of <italic>K</italic> effectively encodes the channel-wise characteristics of original feature map <italic>X</italic> along the channel dimension <italic>C</italic>. The length of each of the <italic>C</italic>&#x02212;dimensional feature vector is <italic>H</italic>&#x000D7;<italic>W</italic>. Meanwhile, each column of <italic>Q</italic>&#x02032; models one of the <italic>s</italic><sup>2</sup>-dimensional feature vectors with the length of <italic>H</italic>&#x000D7;<italic>W</italic>.</p>
<p>Afterwards, in order to explicitly model the interactions between each column of <italic>K</italic> &#x02208; <italic>Q</italic><sup><italic>H</italic>&#x000D7;<italic>W</italic>&#x000D7;<italic>C</italic></sup> and <italic>Q</italic>&#x02032; &#x02208; <italic>R</italic><sup>(<italic>H</italic>&#x000D7;<italic>W</italic>) &#x000D7; <italic>s</italic><sup>2</sup></sup> for all the (<italic>H</italic>&#x000D7;<italic>W</italic>) pixel locations, we employ following operations:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mtext>&#x000A0;</mml:mtext></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>q</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>H</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msup><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mtext>&#x000A0;</mml:mtext></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>q</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000D7;</mml:mo><mml:msup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mtext>&#x000A0;</mml:mtext></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>q</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where <italic>i</italic> &#x0003D; 1, 2, ....., <italic>s</italic><sup>2</sup>, <italic>j</italic> &#x0003D; 1, 2, ...., <italic>C</italic>. Since the number of <italic>query</italic> vectors is <italic>s</italic><sup>2</sup>, <italic>s</italic><sup>2</sup> feature vectors encoded the interactions between all the pixel locations can be thus obtained. The length of each of the feature vector is <italic>C</italic>. We can also rewrite the above operation of dot product form as a form of matrix multiplication:</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mtext>&#x000A0;</mml:mtext></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:msup><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mtext>&#x000A0;</mml:mtext></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:mo>&#x000D7;</mml:mo><mml:msup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mtext>&#x000A0;</mml:mtext></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where <italic>Q</italic>&#x02032;<sup><italic>T</italic></sup> refers to the transpose of matrix <italic>Q</italic>&#x02032;, <italic>S</italic>&#x02032; &#x02208; <italic>R</italic><sup><italic>s</italic><sup>2</sup>&#x000D7;<italic>C</italic></sup> is the union of all the obtained cues about spatial location relation.</p>
<p>Then, the generated two-dimensional <italic>S</italic>&#x02032; &#x02208; <italic>R</italic><sup><italic>s</italic><sup>2</sup>&#x000D7;<italic>C</italic></sup> is reshaped into 3D form <italic>S</italic> &#x02208; <italic>R</italic><sup><italic>s</italic>&#x000D7;<italic>s</italic>&#x000D7;<italic>C</italic></sup>. We then employ a batch normalization operation to modulate <italic>S</italic>, generating the predicted spatially-varying context kernel. The generated kernel effectively encodes the relation cues between pixels of all spatial locations, which can be used to produce spatially-varying weight factor <italic>F</italic> &#x02208; <italic>R</italic><sup><italic>H</italic>&#x000D7;<italic>W</italic>&#x000D7;<italic>C</italic></sup> for all <italic>H</italic>&#x000D7;<italic>W</italic> spatial locations.</p>
<p>In order to fully exploit the information encoded in the spatially-varying context kernel, the depth-wise convolution is applied on the original feature map <italic>X</italic> with the context kernel <italic>S</italic> as the depth-wise convolution kernel. In this way, each channel of <italic>S</italic> is able to modulate one specific channel of <italic>X</italic> in an independent manner. The spatially-varying context guided modulation can thus be implemented. Concretely, as shown in <xref ref-type="fig" rid="F3">Figure 3</xref>, we first split the context kernel <italic>S</italic> &#x02208; <italic>R</italic><sup><italic>s</italic>&#x000D7;<italic>s</italic>&#x000D7;<italic>C</italic></sup> into <italic>C</italic> two-dimensional kernels along the channel dimension. Each of the 2D <italic>C</italic> kernels has a spatial dimension of <italic>s</italic>&#x000D7;<italic>s</italic>. These <italic>C</italic> kernels are then applied on each channel of the original feature map <italic>X</italic> &#x02208; <italic>R</italic><sup><italic>H</italic>&#x000D7;<italic>W</italic>&#x000D7;<italic>C</italic></sup> in an independent manner, generating intermediate feature. A 1 &#x000D7; 1 &#x000D7; 1 convolution is then used to transform the generated intermediate feature map for further feature abstraction. The obtained feature is then processed with one Sigmoid activation function, which produces the spatially-varying weight factor <italic>F</italic> &#x02208; <italic>R</italic><sup><italic>H</italic>&#x000D7;<italic>W</italic>&#x000D7;<italic>C</italic></sup>. Finally, an element-wise multiplication between <italic>M</italic> and <italic>X</italic> is performed to achieve the output feature map, which is then passed through the decoder part for multi-scale feature fusion.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Details of the proposed spatially-varying adaptive pyramid context aggregation module. The adaptive pyramid mechanism aggregates the contextual cues with multi-scale field-of-views via convolution pyramid with multiple atrous rates.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-16-1117134-g0003.tif"/>
</fig>
</sec>
<sec>
<title>3.3. Spatially-varying adaptive pyramid context aggregation module</title>
<sec>
<title>3.3.1. Dilated convolution</title>
<p>Standard convolution is characterized by its property of local receptive field. However, large receptive field is essential for enhancing deep neural network&#x00027;s discriminative feature learning ability. Hence, pooling layer is used after several convolutional layers to enlarge the receptive field. However, the adoption of pooling layer leads to the loss of spatial details and lower-resolution feature map, which is unfavorable for accurate pixel-level keypoint location and matching. Dilated convolution is able to effectively alleviate this challenging issue by sparsifying the standard convolution separated by zero with specific interval (dilation rate), which allows us to enlarge the receptive field without loss of spatial resolution of the feature map.</p>
</sec>
<sec>
<title>3.3.2. Depth-wise dilated convolution</title>
<p>Depth-wise separable convolution transforms the standard convolution into a depth-wise convolution followed by a point-wise convolution. In this way, the computation complexity is thus drastically reduced. Concretely, the depth-wise convolution is applied on each channel of the feature map independently. The point-wise convolution is then used to fuse the output from the depth-wise convolution.</p>
<p>On the basis of the context aggregation module above, a dilation pyramid based context aggregation module is incorporated for further context aggregation of multi-scale field-of-view, as shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. Concretely, with the predicted spatially-varying context kernel <italic>S</italic>, we employ three parallel dilated convolutions with different dilation rates to model contextual cues in a context-adaptive manner. The three different dilation rates are set as 1, 3, 5 in our paper. In this way, a dilation pyramid context aggregation block is obtained.</p>
<p>With these operations, three context kernels (<italic>S</italic>1, <italic>S</italic>2, and <italic>S</italic>3) with different context aggregation fields are obtained. The three context kernels are then applied over the original feature map <italic>X</italic>, leading to three different weight factors <italic>R</italic><sub>1</sub>, <italic>R</italic><sub>2</sub>, and <italic>R</italic><sub>3</sub>. The three generated weight factors are then fused by element-wise sum:</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>R</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>With the final fused weight kernel <italic>R</italic>, similar to the above SCA module, an element-wise multiplication is operated between <italic>R</italic> and <italic>X</italic> to ensure each channel of <italic>R</italic> can independently modulate the corresponding channel of <italic>X</italic>.</p>
</sec>
</sec>
<sec>
<title>3.4. Loss function</title>
<p>In order to supervise the above network to effectively locate and match the keypoints, specifically designed loss functions are utilized.</p>
<sec>
<title>3.4.1. Keypoint location loss</title>
<p>We convert the keypoint location task into a pixel-level binary classification problem. In order to accurately locate the keypoints, the widely used cross-entropy loss is first utilized to supervise the learning of the transformed feature map of the last feature block of the above adaptive pyramid context aggregation network:</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>C</mml:mi><mml:mi>E</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where <italic>y</italic><sub><italic>i</italic></sub> means the label of pixel i (1 and 0 means the keypoint and background, respectively), <italic>p</italic><sub><italic>i</italic></sub> refers to the predicted probability of pixel i to be the keypoint.</p>
<p>We also use the Dice loss for more accurate keypoint location:</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>D</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>e</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi><mml:mo>,</mml:mo><mml:mi>Y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mo>|</mml:mo><mml:mi>P</mml:mi><mml:mo>&#x02229;</mml:mo><mml:mi>Y</mml:mi><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mi>P</mml:mi><mml:mo>|</mml:mo><mml:mo>&#x0002B;</mml:mo><mml:mo>|</mml:mo><mml:mi>Y</mml:mi><mml:mo>|</mml:mo></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where P means the pixel set of the predicted keypoints, Y means the pixel set of ground truth keypoints. |<italic>P</italic>&#x02229;<italic>Y</italic>| refers to the sum of the element-wise production between <italic>P</italic> and <italic>Y</italic>. |<italic>P</italic>|&#x0002B;|<italic>Y</italic>|, |<italic>P</italic>| refers to the sum of all the elements of <italic>P</italic>, |<italic>Y</italic>| refers to the sum of all the elements of <italic>Y</italic>.</p>
</sec>
<sec>
<title>3.4.2. Keypoint matching loss</title>
<p>In order to supervise the network to enhance the discriminative power of learned feature embedding of keypoints, proper keypoint matching loss should be designed. The ideal keypoint matching loss should reduce the gap between matching keypoints and enlarge the gap between non-matching keypoints.</p>
<p>To this end, with the feature map in the last feature block of decoder before generating keypoint detection prediction, we transform this feature map into three-dimensional feature embedding. Thus, every fundus image keypoint has its corresponding one-dimensional feature embedding. Following Huang et al. (<xref ref-type="bibr" rid="B15">2016</xref>) and Opitz et al. (<xref ref-type="bibr" rid="B28">2017</xref>), we set the feature embedding dimension as 512. In this way, our task is to enlarge the distance of feature embedding between non-matching keypoints and narrow the distance of feature embedding between the matching keypoints, leading to compactness between matching keypoints and dispersion between non-matching keypoints. Metric learning mechanism is employed to tackle the above problem in this paper. Concretely, we use the ranking loss to compute the relative distance between the one dimensional feature embedding of every two keypoints in the input image pair.</p>
<sec>
<title>3.4.2.1. Pair-wise ranking loss</title>
<p>This widely used loss is also called contrastive loss. Positive and negative pairs of the one-dimensional feature embedding of keypoints in input image pair are both required for computing the pair-wise ranking loss. One positive pair consists of an anchor keypoint <italic>k</italic><sub><italic>a</italic></sub> and the matching keypoint <italic>k</italic><sub><italic>p</italic></sub>. One negative pair consists of an anchor keypoint and a non-matching keypoint <italic>k</italic><sub><italic>n</italic></sub>. The one-dimensional feature embedding of the anchor keypoint <italic>k</italic><sub><italic>a</italic></sub>, the matching keypoint <italic>k</italic><sub><italic>p</italic></sub> and the non-matching keypoint <italic>k</italic><sub><italic>n</italic></sub> are <italic>f</italic><sub><italic>a</italic></sub>, <italic>f</italic><sub><italic>p</italic></sub>, and <italic>f</italic><sub><italic>n</italic></sub>, respectively. For positive pairs, the aim of the pair-wise ranking loss is to guide the network to learn proper feature embedding with a small distance. On the contrary, for negative pairs, the pair-wise ranking loss aims to supervise the network to learn feature embedding with a large distance. We choose the Euclidian distance as the distance computing function to measure the similarity between the feature embedding. The above operations can be formulated as:</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>L</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable columnalign="right"><mml:mtr><mml:mtd><mml:mi>d</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd><mml:mtd><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>P</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>P</mml:mi><mml:mi>a</mml:mi><mml:mi>i</mml:mi><mml:mi>r</mml:mi><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mi>m</mml:mi><mml:mo>-</mml:mo><mml:mi>d</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd><mml:mtd><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>N</mml:mi><mml:mi>e</mml:mi><mml:mi>g</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>P</mml:mi><mml:mi>a</mml:mi><mml:mi>i</mml:mi><mml:mi>r</mml:mi><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>As shown in the Equation 7, for one positive pair, if the distance between <italic>f</italic><sub><italic>a</italic></sub> and <italic>f</italic><sub><italic>p</italic></sub> are larger than 0, the loss value will also be positive. Hence, the network is guided to reduce the distance to be 0. In this way, this pair-wise ranking loss guides the network to produce similar feature embedding for matching keypoints. On the other hand, for negative pair, when the distance between the feature embedding of the anchor keypoint and negative (non-matching) keypoint is larger than a specific margin threshold, the loss will be 0. When the distance is reduced below the margin value, the loss value will be positive. When the distance between <italic>f</italic><sub><italic>a</italic></sub> and <italic>f</italic><sub><italic>p</italic></sub>, the loss value is the largest value <italic>m</italic>. In this way, the pair-wise ranking loss supervises the network to produce dissimilar feature embedding for non-matching keypoints. When the distance for a negative pair is distant enough (larger than the default threshold), the network will focus on the learning of feature embedding for more difficult pairs.</p>
</sec>
<sec>
<title>3.4.2.2. Triplet ranking loss</title>
<p>Instead of using only one pair of keypoints for every computation of pair-wise ranking loss, the triplet ranking loss considers the relations of a triplet, which consists of an anchor keypoint <italic>k</italic><sub><italic>a</italic></sub>, a positive keypoint <italic>k</italic><sub><italic>p</italic></sub> and a negative keypoint <italic>k</italic><sub><italic>n</italic></sub>. The aim of the triplet ranking loss is to guide the network to produce separable feature embedding: the distance between the feature embedding of the anchor keypoint and negative keypoint <italic>d</italic>(<italic>f</italic><sub><italic>a</italic></sub>, <italic>f</italic><sub><italic>n</italic></sub>) is larger than the distance between the feature embedding of anchor keypoint and the positive keypoint <italic>d</italic>(<italic>f</italic><sub><italic>a</italic></sub>, <italic>r</italic><sub><italic>p</italic></sub>) by a specific margin <italic>m</italic>). The above operations can be rewritten as:</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M9"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>L</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mi>m</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>d</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>d</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>We note that the difference between the pair-wise ranking loss and triplet ranking loss is that pair-wise ranking loss only considers pair of keypoints for one loss computation, however, a triplet of anchor keypoint, positive keypoint and negative keypoint is considered for the triplet ranking loss.</p>
</sec>
<sec>
<title>3.4.2.3. Structured triplet ranking loss</title>
<p>Triplet loss (Weinberger and Saul, <xref ref-type="bibr" rid="B44">2009</xref>; Schroff et al., <xref ref-type="bibr" rid="B33">2015</xref>) is proposed to pull the learned feature embedding of anchor keypoint closer to the positive keypoint than to the negative keypoint by a fixed margin. However, the triplet loss only considers one triplet for every loss computation, neglecting the relations among multiple keypoints. To this end, inspired from Oh Song et al. (<xref ref-type="bibr" rid="B26">2016</xref>); Wang X. et al. (<xref ref-type="bibr" rid="B43">2019</xref>), we propose to employ the structured triplet ranking loss to supervise the feature embedding learning of our network, which explores the structured relationship among multiple keypoints.</p>
<p>Concretely, the structured triplet ranking loss encourages the interaction between more negative keypoints. On the basis of triplet loss, the employed structured triplet ranking loss aims to supervise the learned feature embedding between the anchor keypoint and one positive keypoint is as similar as possible. Moreover, the feature embedding between the anchor keypoint and all negative keypoints as dissimilar as possible. Formally, the structured triplet ranking loss aims to pull the anchor keypoint closer to one positive keypoint than all negative keypoints than a margin <italic>m</italic>.</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M10"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>L</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mn>2</mml:mn><mml:mo>&#x0007C;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>P</mml:mi></mml:mstyle><mml:mo>&#x0007C;</mml:mo></mml:mrow></mml:mfrac><mml:mstyle displaystyle='true'><mml:munder><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02208;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>P</mml:mi></mml:mstyle></mml:mrow></mml:munder><mml:mrow><mml:mo stretchy='false'>[</mml:mo><mml:mi>d</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mstyle><mml:mo>+</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle displaystyle='true'><mml:munder><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>p</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02208;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>N</mml:mi></mml:mstyle></mml:mrow></mml:munder><mml:mi>e</mml:mi></mml:mstyle><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>m</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mi>d</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mi>p</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>+</mml:mo><mml:mstyle displaystyle='true'><mml:munder><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02208;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>N</mml:mi></mml:mstyle></mml:mrow></mml:munder><mml:mi>e</mml:mi></mml:mstyle><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>m</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mi>d</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mi>l</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:msub><mml:mo stretchy='false'>]</mml:mo><mml:mo>+</mml:mo></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where <bold>P</bold> and <bold>N</bold> are the set of positive pairs and negative pairs respectively, <italic>f</italic><sub><italic>i</italic></sub>, <italic>f</italic><sub><italic>p</italic></sub>, <italic>f</italic><sub><italic>j</italic></sub>, and <italic>f</italic><sub><italic>l</italic></sub> refer to the feature embedding of pixel <italic>i</italic>, pixel <italic>p</italic>, pixel <italic>j</italic>, and pixel <italic>l</italic>, respectively. [&#x000B7;]<sub>&#x0002B;</sub> is the hinge function. Illustration of the Pair-wise Ranking loss, Triplet Ranking loss, and Structured Triplet Ranking loss are shown in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Illustration of the <bold>(A)</bold> Pair-wise Ranking loss, <bold>(B)</bold> Triplet Ranking loss, and <bold>(C)</bold> Structured Triplet Ranking loss. Different shapes represent different classes. The blue circle is an anchor. For Pair-wise Ranking loss, the anchor and one positive example or one negative example are considered for every loss computation. For Triplet Ranking loss, the anchor is compared with only one negative example and one positive example. For the Structured Triplet Ranking loss, the anchor is compared with all negative examples.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-16-1117134-g0004.tif"/>
</fig>
</sec>
</sec>
</sec>
<sec>
<title>3.5. Implementation details</title>
<p>The hyperparameters of batch-size, weight decay are set to 1, 1<italic>e</italic>&#x02212;3 respectively. The monmentum is set as 0.9. We use pytorch (Paszke et al., <xref ref-type="bibr" rid="B29">2017</xref>) as the basic implement architecture. The widely used stochastic gradient descent strategy is used for training the proposed model.</p>
</sec>
</sec>
<sec id="s4">
<title>4. Experiments</title>
<p>In this section, we present extensive experiments to validate the proposed model for fundus image registration. First, we show our evaluation dataset and metric. Then we present a detailed analysis of our model on the constructed large-scale dataset.</p>
<sec>
<title>4.1. Datasets and metrics</title>
<sec>
<title>4.1.1. Dataset</title>
<p>Current widely used funds image registration dataset, FIRE, consists of 134 image pairs from 39 patients, which are acquired with Nidek AFC-210 fundus camera. The keypoints of images in FIRE dataset are randomly labeled in a sparse manner. There is not a guarantee that all the vessel branching and crossing points are labeled as keypoints. In this case, these sparse ground-truth keypoint labelings fail to train our proposed model. As a result, a large-scale fundus image registration dataset, which labels all the keypoints in a reliable and stable manner, is required for further research.</p>
<p>To this end, we collect 200 pairs of fundus images under various imaging conditions (illumination, angle etc.) taken from different fundus cameras, such as Canon and RetCam3, as shown in <xref ref-type="table" rid="T1">Table 1</xref>. The constructed dataset is termed as AN-200 dataset. Concretely, 100 high-quality retinal images of 27 adult patients are acquired from Canon. Moreover, the neonatal fundus images are often with low image quality, due to the uncooperative image acquiring process. We collect 100 neonatal fundus images taken from 23 patients with RetCam3 to support various neonatal applications. In addition, different imaging angles and lighting conditions are considered during the construction of the dataset. Example of the fundus images are shown in <xref ref-type="fig" rid="F5">Figure 5</xref>. For every image pair, all the branching and crossing points are labeled as keypoints. All the matched keypoints are then labeled as ground truth matching keypoints. In this way, a reliable and stable fundus image registration dataset is constructed.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Details of our constructed AN-200 dataset.</p></caption>
<table frame="box" rules="all">
<thead><tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="left"><bold>Camera</bold></th>
<th valign="top" align="center"><bold>Number of image pairs</bold></th>
<th valign="top" align="center"><bold>Number of patients</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Adult</td>
<td valign="top" align="left">Canon</td>
<td valign="top" align="center">100</td>
<td valign="top" align="center">27</td>
</tr>
<tr>
<td valign="top" align="left">Neonatus</td>
<td valign="top" align="left">RetCam3</td>
<td valign="top" align="center">100</td>
<td valign="top" align="center">23</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>We collect and label 200 fundus image pairs of adult and neonatus, which are taken from Canon and RetCam3.</p>
</table-wrap-foot>
</table-wrap>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Example of the fundus images from diverse applications, including adult and neonatus patients acquired under good or bad imaging conditions. Moreover, different imaging angles and overlapping areas between the image pairs are also considered.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-16-1117134-g0005.tif"/>
</fig>
</sec>
<sec>
<title>4.1.2. Evaluation metric</title>
<p>First, we choose the widely used FIRE dataset to quantitatively evaluate the proposed method and compare with state of the art methods. Since FIRE dataset only labels part of the crossing and branching points, our model cannot be trained on this dataset. Following Rivas-Villar et al. (<xref ref-type="bibr" rid="B30">2022</xref>), we train the models on the training set of our constructed dataset. The trained models are then evaluated on FIRE dataset with the registration score proposed by Hernandez-Matas et al. (<xref ref-type="bibr" rid="B12">2017</xref>), which calculate the success ratio between the fixed and moving image pairs after the transformation of the moving image with the learned transformation parameters.</p>
<p>Concretely, pixels of moving image are first transformed into the coordinate space of fixed image. We then calculate the averaged distance between the transformed pixels and the ground-truth points of fixed image as the registration error of this image pair. If the registration error is below a threshold, the registration of this image pair is successful. With larger threshold, more image pairs are deemed successful registrations. By varying the threshold from 0 to larger value, the percentage of successful registration pairs enlarges gradually. In this way, we can plot the registration curve, where the X axis corresponds to the setting threshold, the Y axis refers to the percentage of successfully registered images. With the plotting curve, the Area Under Curve (AUC) can be calculated as the final registration score. The original FIRE dataset (Hernandez-Matas et al., <xref ref-type="bibr" rid="B12">2017</xref>) is divided into three sub-datasets based on the overlapping and anatomical similarity between an image pair. The sub-dataset <italic>S</italic> consists of 71 image pairs with more than 75% overlapping and no anatomical differences. The sub-dataset <italic>P</italic> contains 49 image pairs with less than 75% overlapping. Finally, the sub-dataset <italic>A</italic> is composed of 14 image pairs with anatomical differences. Similar to Rivas-Villar et al. (<xref ref-type="bibr" rid="B30">2022</xref>), we calculate the AUC score on the <italic>S</italic>, <italic>P</italic>, and <italic>A</italic> sub-datasets and the whole FIRE dataset.</p>
<p>In addition, we also calculate the AUC value as the registration score on our constructed AN-200 dataset with the same computing manner. Concretely, 60%, 20% and 20% of the original dataset are randomly divided into the training, validation and test set, respectively. The final registraction score is reported on the test set.</p>
</sec>
</sec>
<sec>
<title>4.2. Ablation study on the network architecture</title>
<p>Based on the constructed dataset, in order to obtain better understanding of the proposed network, we evaluate following methods with different network settings. The experimental results are summarized in <xref ref-type="table" rid="T2">Table 2</xref>:</p>
<list list-type="bullet">
<list-item><p>Baseline: We first choose the vanilla encoder-decoder architecture (U-Net) as the backbone network to simultaneously learn the detection of keypoint and the generation of feature embedding, under the supervision of the above cross-entropy loss, Dice loss and the proposed structured triplet ranking loss. As shown in <xref ref-type="table" rid="T2">Table 2</xref>, the Baseline achieves an AUC of 70.5 and 68.1% on AN-200 and FIRE datasets, respectively.</p></list-item>
<list-item><p>Spatially-varying context aggregation network (SCA-Net): Then we enhance the simple U-Net with the proposed spatially-varying context aggregation module. Concretely, over the last stage of the encoder sub-network of U-Net, the generated feature map of encoder sub-network is enhanced with the SCA module. The global contextual cues are thus incorporated. The loss functions are kept the same with the Baseline. The AUC on AC-200 of SCA-Net is 72.2%, and the AUC on FIRE is enlarged to 69.5%. The performance improvement is 1.7 and 1.4%, respectively.</p></list-item>
<list-item><p>Spatially-varying adaptive pyramid context aggregation network (SAPCA-Net): Finally, we test our overall network, SAPCA-Net, by changing the SCA-module with the SAPCA module to incorporate context-adaptive cues. Compared to original U-Net, the SAPCA-Net largely improves the AUC of AD-200 by 2.4%, the AUC of FIRE by 3.0%. Concretely, the AUC of AD-200 is significantly enlarged from 70.5 to 72.9%, and the AUC of FIRE is improved from 68.1 to 71.1%. These results effectively show the effectiveness of the proposed SAPCA module.</p></list-item>
</list>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>The evaluation results of methods with different network settings on AN-200 and FIRE datasets.</p></caption>
<table frame="box" rules="all">
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td valign="top" align="left"><bold>Method</bold></td>
<td valign="top" align="center"><bold>AN-200(%)</bold></td>
<td valign="top" align="center"><bold>FIRE(%)</bold></td>
</tr> <tr>
<td valign="top" align="left">U-Net</td>
<td valign="top" align="center">70.5</td>
<td valign="top" align="center">68.1</td>
</tr> <tr>
<td valign="top" align="left">SCA-Net</td>
<td valign="top" align="center">72.2</td>
<td valign="top" align="center">69.5</td>
</tr> <tr>
<td valign="top" align="left">SAPCA-Net</td>
<td valign="top" align="center"><bold>72.9</bold></td>
<td valign="top" align="center"><bold>71.1</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>The bold values mean the best performance.</p>
</table-wrap-foot>
</table-wrap>
<p>As shown in <xref ref-type="fig" rid="F6">Figure 6</xref>, we plot the curve of the successful registration ratio as the change of different error thresholds. In addition to the above quantitative comparisons, we also show the visualized results of our method. <xref ref-type="fig" rid="F7">Figure 7</xref> demonstrates the visualized keypoint detection and keypoint matching results from two typical scenarios. The last row also shows the final fused results with the matching keypoints. The first column of <xref ref-type="fig" rid="F7">Figure 7</xref> shows the ground truth keypoint detection and fused result. As shown in <xref ref-type="fig" rid="F7">Figure 7</xref>, the baseline method is able to effectively locate and match keypoints. However, there exist a number of wrong keypoint matching results. The SCA-Net is able to remove some false positive predictions, leading to better keypoint matching result. Finally, the SAPCA-Net further removes more false positive keypoint matching predictions. Meanwhile, the number of true keypoint matching is also increased. As a result, the final fused result with the matching keypoints generated by the SAPCA-Net is visually better than other methods. These qualitative comparisons further demonstrate the effectiveness of the proposed network architecture.</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>The registration success with different error thresholds for the model with different network settings on FIRE dataset.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-16-1117134-g0006.tif"/>
</fig>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>Example of the keypoint detection and matching results of normal adult and neonatal fundus images. We also show the fused image with the matching keypoints.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-16-1117134-g0007.tif"/>
</fig>
</sec>
<sec>
<title>4.3. Ablation study on the loss function</title>
<p>On the basis of the above best performing SAPCA-Net, we also conduct further ablation study for further understanding of the loss function. We evaluate the SAPCA-Net with following different loss functions, the results are summarized in <xref ref-type="table" rid="T3">Table 3</xref>:</p>
<list list-type="bullet">
<list-item><p>SAPCA-net-pairwise: We first replace the keypoint matching loss function of SAPCA-Net with the simple pairwise ranking loss. Pairwise ranking loss guides the SAPCA-Net to learn the pairwise relationship between the feature embedding of the anchor keypoint and one positive/negative keypoint. As shown in <xref ref-type="table" rid="T3">Table 3</xref>, the SAPCA-Net-Pairwise achieves the AUC of 71.4 and 69.7% on AN-200 and FIRE, respectively.</p></list-item>
<list-item><p>SAPCA-net-triplet: Then we replace the keypoint matching loss function with the triplet ranking loss. The triplet loss helps the network to pull the anchor point closer to the similar keypoint than the dissimilar one by a margin. The AUC of SAPCA-Net-Triplet on AN-200 is 72.2%, and the AUC on FIRE is improved to 70.3%.</p></list-item>
<list-item><p>SAPCA-net-structured-triplet: We further replace the keypoint matching loss function with structured triplet ranking loss. The structured triplet ranking loss supervise the network to learn the structured relationship among multiple keypoints. Compared to original pair-wise ranking loss, the AUC of AN-200 is enlarged from 71.4 to 72.9%, and the AUC of FIRE is improved from 69.7 to 71.1%. These results effectively show the effectiveness of the employed structured triplet ranking loss.</p></list-item>
</list>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Ablation study on the loss function.</p></caption>
<table frame="box" rules="all">
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td valign="top" align="left"><bold>Method</bold></td>
<td valign="top" align="center"><bold>AN-200(%)</bold></td>
<td valign="top" align="center"><bold>FIRE(%)</bold></td>
</tr> <tr>
<td valign="top" align="left">SAPCA-Net-Pairwise</td>
<td valign="top" align="center">71.4</td>
<td valign="top" align="center">69.7</td>
</tr> <tr>
<td valign="top" align="left">SAPCA-Net-Triplet</td>
<td valign="top" align="center">72.2</td>
<td valign="top" align="center">70.3</td>
</tr> <tr>
<td valign="top" align="left">SAPCA-Net-Structured-Triplet</td>
<td valign="top" align="center"><bold>72.9</bold></td>
<td valign="top" align="center"><bold>71.1</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>With the structured triplet loss, SAPCA-Net-Structured-Triplet achieves the best result. The bold values mean the best performance.</p>
</table-wrap-foot>
</table-wrap>
<p>Among the SAPCA-Net with the above three different loss functions, the SAPCA-Net-Structured-Triplet achieves significantly better results, which effectively demonstrates the superiority of the structured triplet ranking loss for the learning of matching keypoints. The change curve of registration success ratio under different error thresholds is shown in <xref ref-type="fig" rid="F8">Figure 8</xref>.</p>
<fig id="F8" position="float">
<label>Figure 8</label>
<caption><p>The curve of successful registration percentage under different error thresholds for the model with different loss functions on FIRE dataset.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-16-1117134-g0008.tif"/>
</fig>
</sec>
<sec>
<title>4.4. Comparison to state-of-arts</title>
<p>In order to compare our proposed best-performing SAPCA-Net with state of the art methods, the widely used FIRE dataset is employed for evaluation. We first focus on the deep learning based methods. As shown in <xref ref-type="table" rid="T4">Table 4</xref>, compared to previous two-stage UNet &#x0002B; RANSAC (Rivas-Villar et al., <xref ref-type="bibr" rid="B30">2022</xref>), our end-to-end registration method achieves consistently better results on the S, P, A sub-datasets and the whole FIRE dataset. Concretely, on the four dataset settings, our SAPCA-Net achieves the registration score of 93.9, 36.2, 71.9, and 71.1%, significantly outperforming UNet &#x0002B; RANSAC by 3.1, 6.9, 5.9, and 5.4%, respectively. Moreover, our model accomplishes the two steps of keypoint detection and matching with a single network. However, for previous UNet &#x0002B; RANSAC model, the keypoint detection is first accomplished by a U-Net, which is followed by traditional RANSAC (Fischler and Bolles, <xref ref-type="bibr" rid="B8">1981</xref>) for the keypoint matching step. In this way, the execution time of our proposed SAPCA-Net is much shorter.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Comparison to state-of-arts on FIRE dataset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center"><bold>S</bold></th>
<th valign="top" align="center"><bold>P</bold></th>
<th valign="top" align="center"><bold>A</bold></th>
<th valign="top" align="center"><bold>FIRE</bold></th>
<th valign="top" align="center"><bold>Execution time</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">SIFT &#x0002B; WGTM (Lowe, <xref ref-type="bibr" rid="B22">2004</xref>)</td>
<td valign="top" align="center">83.7</td>
<td valign="top" align="center">54.4</td>
<td valign="top" align="center">40.7</td>
<td valign="top" align="center">68.5</td>
<td valign="top" align="center">&#x02013;</td>
</tr> <tr>
<td valign="top" align="left">GDB-ICP (Yang et al., <xref ref-type="bibr" rid="B45">2007</xref>)</td>
<td valign="top" align="center">81.4</td>
<td valign="top" align="center">30.3</td>
<td valign="top" align="center">30.3</td>
<td valign="top" align="center">57.6</td>
<td valign="top" align="center">19</td>
</tr> <tr>
<td valign="top" align="left">Harris-PIIFD (Yang et al., <xref ref-type="bibr" rid="B45">2007</xref>)</td>
<td valign="top" align="center">90.0</td>
<td valign="top" align="center">9.0</td>
<td valign="top" align="center">44.3</td>
<td valign="top" align="center">55.3</td>
<td valign="top" align="center">13</td>
</tr> <tr>
<td valign="top" align="left">SURF &#x0002B; WGTM (Bay et al., <xref ref-type="bibr" rid="B1">2008</xref>)</td>
<td valign="top" align="center">83.5</td>
<td valign="top" align="center">6.1</td>
<td valign="top" align="center">6.9</td>
<td valign="top" align="center">47.2</td>
<td valign="top" align="center">&#x02013;</td>
</tr> <tr>
<td valign="top" align="left">ED-DB-ICP (Tsai et al., <xref ref-type="bibr" rid="B39">2009</xref>)</td>
<td valign="top" align="center">60.4</td>
<td valign="top" align="center">44.1</td>
<td valign="top" align="center">49.7</td>
<td valign="top" align="center">55.3</td>
<td valign="top" align="center">44</td>
</tr> <tr>
<td valign="top" align="left">RIR-BS (Chen et al., <xref ref-type="bibr" rid="B4">2011</xref>)</td>
<td valign="top" align="center">77.2</td>
<td valign="top" align="center">0.49</td>
<td valign="top" align="center">12.4</td>
<td valign="top" align="center">44.0</td>
<td valign="top" align="center">-</td>
</tr> <tr>
<td valign="top" align="left">ATS-RGN (Serradell et al., <xref ref-type="bibr" rid="B34">2014</xref>)</td>
<td valign="top" align="center">36.9</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">14.7</td>
<td valign="top" align="center">21.1</td>
<td valign="top" align="center">-</td>
</tr> <tr>
<td valign="top" align="left">EyeSLAM (Braun et al., <xref ref-type="bibr" rid="B2">2018</xref>)</td>
<td valign="top" align="center">30.8</td>
<td valign="top" align="center">22.4</td>
<td valign="top" align="center">26.9</td>
<td valign="top" align="center">27.3</td>
<td valign="top" align="center">7</td>
</tr> <tr>
<td valign="top" align="left">GFEMR (Wang J. et al., <xref ref-type="bibr" rid="B41">2019</xref>)</td>
<td valign="top" align="center">81.2</td>
<td valign="top" align="center">60.7</td>
<td valign="top" align="center">47.4</td>
<td valign="top" align="center">70.2</td>
<td valign="top" align="center">10</td>
</tr> <tr>
<td valign="top" align="left">RIFT &#x0002B; NTG (Zhou et al., <xref ref-type="bibr" rid="B47">2022</xref>)</td>
<td valign="top" align="center">90.7</td>
<td valign="top" align="center">51.2</td>
<td valign="top" align="center">81.0</td>
<td valign="top" align="center">71.7</td>
<td valign="top" align="center">-</td>
</tr> <tr>
<td valign="top" align="left">VOTUS (Motta et al., <xref ref-type="bibr" rid="B23">2019</xref>)</td>
<td valign="top" align="center">93.4</td>
<td valign="top" align="center">67.2</td>
<td valign="top" align="center">68.1</td>
<td valign="top" align="center">81.2</td>
<td valign="top" align="center">106</td>
</tr> <tr>
<td valign="top" align="left">REMPE (Hernandez-Matas et al., <xref ref-type="bibr" rid="B11">2020</xref>)</td>
<td valign="top" align="center">95.8</td>
<td valign="top" align="center">54.2</td>
<td valign="top" align="center">66.0</td>
<td valign="top" align="center">77.3</td>
<td valign="top" align="center">198</td>
</tr> <tr>
<td valign="top" align="left">U-Net &#x0002B; RANSAC (Rivas-Villar et al., <xref ref-type="bibr" rid="B30">2022</xref>)</td>
<td valign="top" align="center">90.8</td>
<td valign="top" align="center">29.3</td>
<td valign="top" align="center">66.0</td>
<td valign="top" align="center">65.7</td>
<td valign="top" align="center">0.65</td>
</tr>
<tr>
<td valign="top" align="left">Our SAPCA-Net</td>
<td valign="top" align="center">93.9</td>
<td valign="top" align="center">36.2</td>
<td valign="top" align="center">71.9</td>
<td valign="top" align="center">71.1</td>
<td valign="top" align="center">0.32</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Then, we compare our method with traditional registration methods. As shown in <xref ref-type="table" rid="T4">Table 4</xref>, our SAPCA-Net obtains the best registration score on the A sub-dataset, by achieving 71.1% AUC. This result is 3.8% better than previous best performing VOTUS. On the S sub-dataset, our method obtains the registration score of 93.9%, slightly better than VOTUS, while is 1.9% lower than the REMPE. On the whole FIRE dataset, our method outperforms most of the traditional methods. Although VOTUS and REMPE achieve better registration scores than our SAPCA-Net, the execution time of these two methods are two orders of magnitude slower than our method. Concretely, the execution time of our method is only 0.32s, which shows significant advantage compared to the VOTUS (106s) and REMPE (198s). This is a big advantage for applications in clinical scenarios.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="s5">
<title>5. Conclusion</title>
<p>Current deep learning based image registration methods directly learn to align the geometric transformation or the dense displacement vector field between the input image pair. These previous modeling paradigms fail to achieve keypoint detection and registration results in a reliable and stable way. To this end, in this paper, we aim to tackle this challenging issue. First, considering that the vessel crossing and branching points can reliably and stably characterize the key components for fundus image, a single network is employed to simultaneously learn to detect and match all the crossing and branching points of the input image pair in an end-to-end manner. Moreover, a spatially-varying adaptive pyramid context aggregation network is proposed to aggregate contextual cues in multi-scale field-of-view, which are much beneficial for accurate keypoint detection and matching. Furthermore, a structured triplet ranking loss is employed to guide the learning of similar feature embedding for matching keypoint and dissimilar feature embedding for non-matching keypoints. The proposed model is trained on a new constructed large-scale dataset with well-labeled ground-truths. Both quantitative and qualitative results show the effectiveness of the proposed method.</p>
</sec>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>JX and YC pointed out the problem of current methods and provided new solution. YC and PS designed the dataset constructing scheme. JX, YC, PS, LD, RS, and ZY collected and labeled the dataset. YC, PS, and DZ cleaned the dataset. JX, KY, LD, DZ, RS, and ZY performed the experiments. JX, YC, and PS evaluated the experimental results. KY wrote the first draft of the manuscript. All the authors revised the manuscript, contributed to the article, and approved the submitted version. All the authors approve the final version to be published and agree to be accountable for all aspects of the work in ensuring that questions related to the accuracy or integrity of any part of the work are appropriately investigated and resolved.</p>
</sec>
</body>
<back>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>This work was supported by the Sichuan Provincial People&#x00027;s Hospital Fund Project No. 2021LY15 and the Chengdu Science and Technology Bureau Project No. 2021-YF05-00498-SN.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>KY, LD, DZ, RS, and ZY were employed by the company Beijing Zhizhen Internet Technology Co. Ltd. The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bay</surname> <given-names>H.</given-names></name> <name><surname>Ess</surname> <given-names>A.</given-names></name> <name><surname>Tuytelaars</surname> <given-names>T.</given-names></name> <name><surname>Van Gool</surname> <given-names>L.</given-names></name></person-group> (<year>2008</year>). <article-title>Speeded-up robust features (surf)</article-title>. <source>Comput. Vis. Image Understand</source>. <volume>110</volume>, <fpage>346</fpage>&#x02013;<lpage>359</lpage>. <pub-id pub-id-type="doi">10.1016/j.cviu.2007.09.014</pub-id></citation>
</ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Braun</surname> <given-names>D.</given-names></name> <name><surname>Yang</surname> <given-names>S.</given-names></name> <name><surname>Martel</surname> <given-names>J. N.</given-names></name> <name><surname>Riviere</surname> <given-names>C. N.</given-names></name> <name><surname>Becker</surname> <given-names>B. C.</given-names></name></person-group> (<year>2018</year>). <article-title>Eyeslam: Real-time simultaneous localization and mapping of retinal vessels during intraocular microsurgery</article-title>. <source>Int. J. Med. Rob. Comput. Assist. Surg</source>. 14, e1848. <pub-id pub-id-type="doi">10.1002/rcs.1848</pub-id><pub-id pub-id-type="pmid">28719002</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cao</surname> <given-names>X.</given-names></name> <name><surname>Yang</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Nie</surname> <given-names>D.</given-names></name> <name><surname>Kim</surname> <given-names>M.</given-names></name> <name><surname>Wang</surname> <given-names>Q.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>&#x0201C;Deformable image registration based on similarity-steered CNN regression,&#x0201D;</article-title> in <source>International Conference on Medical Image Computing and Computer-Assisted Intervention</source> (<publisher-loc>Qu&#x000E9;bec City, QC</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>300</fpage>&#x02013;<lpage>308</lpage>.<pub-id pub-id-type="pmid">29250613</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>L.</given-names></name> <name><surname>Xiang</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name></person-group> (<year>2011</year>). <article-title>&#x0201C;Retinal image registration using bifurcation structures,&#x0201D;</article-title> in <source>IEEE International Conference on Image Processing</source> (<publisher-loc>Brussels</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>2169</fpage>&#x02013;<lpage>2172</lpage>.<pub-id pub-id-type="pmid">31754981</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chopra</surname> <given-names>S.</given-names></name> <name><surname>Hadsell</surname> <given-names>R.</given-names></name> <name><surname>LeCun</surname> <given-names>Y.</given-names></name></person-group> (<year>2005</year>). <article-title>&#x0201C;Learning a similarity metric discriminatively, with application to face verification,&#x0201D;</article-title> in <source>2005 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR&#x00027;05), Vol. 1</source> (<publisher-loc>San Diego, CA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>539</fpage>&#x02013;<lpage>546</lpage>.</citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cui</surname> <given-names>Y.</given-names></name> <name><surname>Zhou</surname> <given-names>F.</given-names></name> <name><surname>Lin</surname> <given-names>Y.</given-names></name> <name><surname>Belongie</surname> <given-names>S.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Fine-grained categorization and dataset bootstrapping using deep metric learning with humans in the loop,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Las Vegas, NV</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1153</fpage>&#x02013;<lpage>1162</lpage>.</citation>
</ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Deng</surname> <given-names>K.</given-names></name> <name><surname>Tian</surname> <given-names>J.</given-names></name> <name><surname>Zheng</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Dai</surname> <given-names>X.</given-names></name> <name><surname>Xu</surname> <given-names>M.</given-names></name></person-group> (<year>2010</year>). <article-title>Retinal fundus image registration via vascular structure graph matching</article-title>. <source>Int. J. Biomed. Imaging</source> <volume>2010</volume>, <fpage>906067</fpage>. <pub-id pub-id-type="doi">10.1155/2010/906067</pub-id><pub-id pub-id-type="pmid">20871853</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fischler</surname> <given-names>M. A.</given-names></name> <name><surname>Bolles</surname> <given-names>R. C.</given-names></name></person-group> (<year>1981</year>). <article-title>Random sample consensus: a paradigm for model fitting with applications to image analysis and automated cartography</article-title>. <source>Commun. ACM</source> <volume>24</volume>, <fpage>381</fpage>&#x02013;<lpage>395</lpage>. <pub-id pub-id-type="doi">10.1145/358669.358692</pub-id></citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hadsell</surname> <given-names>R.</given-names></name> <name><surname>Chopra</surname> <given-names>S.</given-names></name> <name><surname>LeCun</surname> <given-names>Y.</given-names></name></person-group> (<year>2006</year>). <article-title>&#x0201C;Dimensionality reduction by learning an invariant mapping,&#x0201D;</article-title> in <source>2006 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR&#x00027;06), Vol. 2</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1735</fpage>&#x02013;<lpage>1742</lpage>.<pub-id pub-id-type="pmid">33981127</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Ren</surname> <given-names>S.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Deep residual learning for image recognition,&#x0201D;</article-title> in <source>IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Las Vegas, NV</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>770</fpage>&#x02013;<lpage>778</lpage>.<pub-id pub-id-type="pmid">32166560</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hernandez-Matas</surname> <given-names>C.</given-names></name> <name><surname>Zabulis</surname> <given-names>X.</given-names></name> <name><surname>Argyros</surname> <given-names>A. A.</given-names></name></person-group> (<year>2020</year>). <article-title>Rempe: registration of retinal images through eye modelling and pose estimation</article-title>. <source>IEEE J. Biomed. Health Inform</source>. <volume>24</volume>, <fpage>3362</fpage>&#x02013;<lpage>3373</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2020.2984483</pub-id><pub-id pub-id-type="pmid">32248134</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hernandez-Matas</surname> <given-names>C.</given-names></name> <name><surname>Zabulis</surname> <given-names>X.</given-names></name> <name><surname>Triantafyllou</surname> <given-names>A.</given-names></name> <name><surname>Anyfanti</surname> <given-names>P.</given-names></name> <name><surname>Douma</surname> <given-names>S.</given-names></name> <name><surname>Argyros</surname> <given-names>A. A.</given-names></name></person-group> (<year>2017</year>). <article-title>Fire: fundus image registration dataset</article-title>. <source>Model. Artif. Intell. Ophthalmol</source>. <volume>1</volume>, <fpage>16</fpage>&#x02013;<lpage>28</lpage>. <pub-id pub-id-type="doi">10.35119/maio.v1i4.42</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Hershey</surname> <given-names>J. R.</given-names></name> <name><surname>Chen</surname> <given-names>Z.</given-names></name> <name><surname>Le Roux</surname> <given-names>J.</given-names></name> <name><surname>Watanabe</surname> <given-names>S.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Deep clustering: Discriminative embeddings for segmentation and separation,&#x0201D;</article-title> in <source>2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</source> (<publisher-loc>Shanghai</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>31</fpage>&#x02013;<lpage>35</lpage>.</citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hill</surname> <given-names>D. L.</given-names></name> <name><surname>Batchelor</surname> <given-names>P. G.</given-names></name> <name><surname>Holden</surname> <given-names>M.</given-names></name> <name><surname>Hawkes</surname> <given-names>D. J.</given-names></name></person-group> (<year>2001</year>). <article-title>Medical image registration</article-title>. <source>Phys. Med. Biol</source>. 46, R1. <pub-id pub-id-type="doi">10.1088/0031-9155/46/3/201</pub-id><pub-id pub-id-type="pmid">11277237</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>C.</given-names></name> <name><surname>Loy</surname> <given-names>C. C.</given-names></name> <name><surname>Tang</surname> <given-names>X.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Local similarity-aware deep feature embedding,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems, Vol. 29</source> (<publisher-loc>Barcelona</publisher-loc>).</citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jie</surname> <given-names>H.</given-names></name> <name><surname>Shen</surname> <given-names>L.</given-names></name> <name><surname>Samuel</surname> <given-names>A.</given-names></name> <name><surname>Gang</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Squeeze-and-excitation networks,&#x0201D;</article-title> in <source>IEEE Transactions on Pattern Analysis and Machine Intelligence</source> (<publisher-loc>Salt Lake City, UT</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krebs</surname> <given-names>J.</given-names></name> <name><surname>Mansi</surname> <given-names>T.</given-names></name> <name><surname>Delingette</surname> <given-names>H.</given-names></name> <name><surname>Zhang</surname> <given-names>L.</given-names></name> <name><surname>Ghesu</surname> <given-names>F. C.</given-names></name> <name><surname>Miao</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>&#x0201C;Robust non-rigid registration through agent-based action learning,&#x0201D;</article-title> in <source>International Conference on Medical Image Computing and Computer-Assisted Intervention</source> (<publisher-loc>Qu&#x000E9;bec City, QC</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>344</fpage>&#x02013;<lpage>352</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krizhevsky</surname> <given-names>A.</given-names></name> <name><surname>Sutskever</surname> <given-names>I.</given-names></name> <name><surname>Hinton</surname> <given-names>G. E.</given-names></name></person-group> (<year>2012</year>). <article-title>&#x0201C;Imagenet classification with deep convolutional neural networks,&#x0201D;</article-title> in <source>NIPS</source> (<publisher-loc>Nevada</publisher-loc>).</citation>
</ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Law</surname> <given-names>M. T.</given-names></name> <name><surname>Urtasun</surname> <given-names>R.</given-names></name> <name><surname>Zemel</surname> <given-names>R. S.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Deep spectral clustering learning,&#x0201D;</article-title> in <source>International Conference on Machine Learning</source> (<publisher-loc>Lugano</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>1985</fpage>&#x02013;<lpage>1994</lpage>.<pub-id pub-id-type="pmid">34648435</pub-id></citation></ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Fan</surname> <given-names>Y.</given-names></name></person-group> (<year>2017</year>). <article-title>Non-rigid image registration using fully convolutional networks with deep self-supervision</article-title>. <source>arXiv preprint</source> arXiv:1709.00799. <pub-id pub-id-type="doi">10.1109/ISBI.2018.8363757</pub-id><pub-id pub-id-type="pmid">30079127</pub-id></citation></ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>He</surname> <given-names>J.</given-names></name> <name><surname>Qiao</surname> <given-names>Y.</given-names></name> <name><surname>Ren</surname> <given-names>J. S.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Learning to predict context-adaptive convolution for semantic segmentation,&#x0201D;</article-title> in <source>European Conference on Computer Vision</source> (<publisher-loc>Glasgow</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>769</fpage>&#x02013;<lpage>786</lpage>.</citation>
</ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lowe</surname> <given-names>D. G.</given-names></name></person-group> (<year>2004</year>). <article-title>Distinctive image features from scale-invariant keypoints</article-title>. <source>Int. J. Comput. Vis</source>. <volume>60</volume>, <fpage>91</fpage>&#x02013;<lpage>110</lpage>. <pub-id pub-id-type="doi">10.1023/B:VISI.0000029664.99615.94</pub-id></citation>
</ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Motta</surname> <given-names>D.</given-names></name> <name><surname>Casaca</surname> <given-names>W.</given-names></name> <name><surname>Paiva</surname> <given-names>A.</given-names></name></person-group> (<year>2019</year>). <article-title>Vessel optimal transport for automated alignment of retinal fundus images</article-title>. <source>IEEE Trans. Image Process</source>. <volume>28</volume>, <fpage>6154</fpage>&#x02013;<lpage>6168</lpage>. <pub-id pub-id-type="doi">10.1109/TIP.2019.2925287</pub-id><pub-id pub-id-type="pmid">31283507</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Movshovitz-Attias</surname> <given-names>Y.</given-names></name> <name><surname>Toshev</surname> <given-names>A.</given-names></name> <name><surname>Leung</surname> <given-names>T. K.</given-names></name> <name><surname>Ioffe</surname> <given-names>S.</given-names></name> <name><surname>Singh</surname> <given-names>S.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;No fuss distance metric learning using proxies,&#x0201D;</article-title> in <source>Proceedings of the IEEE International Conference on Computer Vision</source> (<publisher-loc>Venice</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>360</fpage>&#x02013;<lpage>368</lpage>.</citation>
</ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oh Song</surname> <given-names>H.</given-names></name> <name><surname>Jegelka</surname> <given-names>S.</given-names></name> <name><surname>Rathod</surname> <given-names>V.</given-names></name> <name><surname>Murphy</surname> <given-names>K.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Deep metric learning via facility location,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Honolulu, HI</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>5382</fpage>&#x02013;<lpage>5390</lpage>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oh Song</surname> <given-names>H.</given-names></name> <name><surname>Xiang</surname> <given-names>Y.</given-names></name> <name><surname>Jegelka</surname> <given-names>S.</given-names></name> <name><surname>Savarese</surname> <given-names>S.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Deep metric learning via lifted structured feature embedding,&#x0201D;</article-title> in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source> (<publisher-loc>Las Vegas, NV</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>4004</fpage>&#x02013;<lpage>4012</lpage>.</citation>
</ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oliveira</surname> <given-names>F. P.</given-names></name> <name><surname>Tavares</surname> <given-names>J. M. R.</given-names></name></person-group> (<year>2014</year>). <article-title>Medical image registration: a review</article-title>. <source>Comput. Methods Biomech. Biomed. Eng</source>. <volume>17</volume>, <fpage>73</fpage>&#x02013;<lpage>93</lpage>. <pub-id pub-id-type="doi">10.1080/10255842.2012.670855</pub-id><pub-id pub-id-type="pmid">22435355</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Opitz</surname> <given-names>M.</given-names></name> <name><surname>Waltner</surname> <given-names>G.</given-names></name> <name><surname>Possegger</surname> <given-names>H.</given-names></name> <name><surname>Bischof</surname> <given-names>H.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Bier-boosting independent embeddings robustly,&#x0201D;</article-title> in <source>Proceedings of the IEEE International Conference on Computer Vision</source> (<publisher-loc>Venice</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>5189</fpage>&#x02013;<lpage>5198</lpage>.<pub-id pub-id-type="pmid">29994466</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Paszke</surname> <given-names>A.</given-names></name> <name><surname>Gross</surname> <given-names>S.</given-names></name> <name><surname>Chintala</surname> <given-names>S.</given-names></name> <name><surname>Chanan</surname> <given-names>G.</given-names></name> <name><surname>Yang</surname> <given-names>E.</given-names></name> <name><surname>DeVito</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>&#x0201C;Automatic differentiation in pytorch,&#x0201D;</article-title> in <source>ICLR</source>.</citation>
</ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rivas-Villar</surname> <given-names>D.</given-names></name> <name><surname>Hervella</surname> <given-names>&#x000C1;. S.</given-names></name> <name><surname>Rouco</surname> <given-names>J.</given-names></name> <name><surname>Novo</surname> <given-names>J.</given-names></name></person-group> (<year>2022</year>). <article-title>Color fundus image registration using a learning-based domain-specific landmark detection methodology</article-title>. <source>Comput. Biol. Med</source>. 140, 105101. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2021.105101</pub-id><pub-id pub-id-type="pmid">34875412</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Roh&#x000E9;</surname> <given-names>M.-M.</given-names></name> <name><surname>Datar</surname> <given-names>M.</given-names></name> <name><surname>Heimann</surname> <given-names>T.</given-names></name> <name><surname>Sermesant</surname> <given-names>M.</given-names></name> <name><surname>Pennec</surname> <given-names>X.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Svf-net: learning deformable image registration using shape matching,&#x0201D;</article-title> in <source>International Conference on Medical Image Computing and Computer-Assisted Intervention</source> (<publisher-loc>Qu&#x000E9;bec City, QC</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>266</fpage>&#x02013;<lpage>274</lpage>.</citation>
</ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ronneberger</surname> <given-names>O.</given-names></name> <name><surname>Fischer</surname> <given-names>P.</given-names></name> <name><surname>Brox</surname> <given-names>T.</given-names></name></person-group> (<year>2015</year>). U-Net: &#x0201C;Convolutional networks for biomedical image segmentation,&#x0201D; in <italic>International Conference on Medical Image Computing and Computer-Assisted Intervention</italic> (Munich).</citation>
</ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schroff</surname> <given-names>F.</given-names></name> <name><surname>Kalenichenko</surname> <given-names>D.</given-names></name> <name><surname>Philbin</surname> <given-names>J.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Facenet: a unified embedding for face recognition and clustering,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Boston, MA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>815</fpage>&#x02013;<lpage>823</lpage>.</citation>
</ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Serradell</surname> <given-names>E.</given-names></name> <name><surname>Pinheiro</surname> <given-names>M. A.</given-names></name> <name><surname>Sznitman</surname> <given-names>R.</given-names></name> <name><surname>Kybic</surname> <given-names>J.</given-names></name> <name><surname>Moreno-Noguer</surname> <given-names>F.</given-names></name> <name><surname>Fua</surname> <given-names>P.</given-names></name></person-group> (<year>2014</year>). <article-title>Non-rigid graph registration using active testing search</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>37</volume>, <fpage>625</fpage>&#x02013;<lpage>638</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2014.2343235</pub-id><pub-id pub-id-type="pmid">26353266</pub-id></citation></ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Simonyan</surname> <given-names>K.</given-names></name> <name><surname>Zisserman</surname> <given-names>A.</given-names></name></person-group> (<year>2014</year>). <article-title>&#x0201C;Very deep convolutional networks for large-scale image recognition,&#x0201D;</article-title> in <source>ICLR</source> (<publisher-loc>Banff, AB</publisher-loc>).</citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sohn</surname> <given-names>K.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Improved deep metric learning with multi-class n-pair loss objective,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems, Vol. 29</source> (<publisher-loc>Barcelona</publisher-loc>).</citation>
</ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sokooti</surname> <given-names>H.</given-names></name> <name><surname>Vos</surname> <given-names>B.d</given-names></name> <name><surname>Berendsen</surname> <given-names>F.</given-names></name> <name><surname>Lelieveldt</surname> <given-names>B. P.</given-names></name> <name><surname>I&#x00161;gum</surname> <given-names>I.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>&#x0201C;Nonrigid image registration using multi-scale 3D convolutional neural networks,&#x0201D;</article-title> in <source>International Conference on Medical Image Computing and Computer-Assisted Intervention</source> (<publisher-loc>Qu&#x000E9;bec City, QC</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>232</fpage>&#x02013;<lpage>239</lpage>.</citation>
</ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sotiras</surname> <given-names>A.</given-names></name> <name><surname>Davatzikos</surname> <given-names>C.</given-names></name> <name><surname>Paragios</surname> <given-names>N.</given-names></name></person-group> (<year>2013</year>). <article-title>Deformable medical image registration: a survey</article-title>. <source>IEEE Trans. Med. Imaging</source> <volume>32</volume>, <fpage>1153</fpage>&#x02013;<lpage>1190</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2013.2265603</pub-id><pub-id pub-id-type="pmid">23739795</pub-id></citation></ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tsai</surname> <given-names>C.-L.</given-names></name> <name><surname>Li</surname> <given-names>C.-Y.</given-names></name> <name><surname>Yang</surname> <given-names>G.</given-names></name> <name><surname>Lin</surname> <given-names>K.-S.</given-names></name></person-group> (<year>2009</year>). <article-title>The edge-driven dual-bootstrap iterative closest point algorithm for registration of multimodal fluorescein angiogram sequence</article-title>. <source>IEEE Trans. Med. Imaging</source> <volume>29</volume>, <fpage>636</fpage>&#x02013;<lpage>649</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2009.2030324</pub-id><pub-id pub-id-type="pmid">19709965</pub-id></citation></ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vos</surname> <given-names>B. D. D.</given-names></name> <name><surname>Berendsen</surname> <given-names>F. F.</given-names></name> <name><surname>Viergever</surname> <given-names>M. A.</given-names></name> <name><surname>Staring</surname> <given-names>M.</given-names></name> <name><surname>I&#x00161;gum</surname> <given-names>I.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;End-to-end unsupervised deformable image registration with a convolutional neural network,&#x0201D;</article-title> in <source>Deep Learning in Medical Image Analysis and Multimodal Learning for Clinical Decision Support</source> (<publisher-loc>Qu&#x000E9;bec City, QC</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>204</fpage>&#x02013;<lpage>212</lpage>.</citation>
</ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Xu</surname> <given-names>H.</given-names></name> <name><surname>Zhang</surname> <given-names>S.</given-names></name> <name><surname>Mei</surname> <given-names>X.</given-names></name> <name><surname>Huang</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Gaussian field estimator with manifold regularization for retinal image registration</article-title>. <source>Signal Process</source>. <volume>157</volume>, <fpage>225</fpage>&#x02013;<lpage>235</lpage>. <pub-id pub-id-type="doi">10.1016/j.sigpro.2018.12.004</pub-id></citation>
</ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Song</surname> <given-names>Y.</given-names></name> <name><surname>Leung</surname> <given-names>T.</given-names></name> <name><surname>Rosenberg</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Philbin</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>&#x0201C;Learning fine-grained image similarity with deep ranking,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Columbus, OH</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1386</fpage>&#x02013;<lpage>1393</lpage>.<pub-id pub-id-type="pmid">33760730</pub-id></citation></ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Hua</surname> <given-names>Y.</given-names></name> <name><surname>Kodirov</surname> <given-names>E.</given-names></name> <name><surname>Hu</surname> <given-names>G.</given-names></name> <name><surname>Garnier</surname> <given-names>R.</given-names></name> <name><surname>Robertson</surname> <given-names>N. M.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Ranked list loss for deep metric learning,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Long Beach, CA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>5207</fpage>&#x02013;<lpage>5216</lpage>.<pub-id pub-id-type="pmid">33760730</pub-id></citation></ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Weinberger</surname> <given-names>K. Q.</given-names></name> <name><surname>Saul</surname> <given-names>L. K.</given-names></name></person-group> (<year>2009</year>). <article-title>Distance metric learning for large margin nearest neighbor classification</article-title>. <source>J. Mach. Learn. Res</source>. <volume>10</volume>, <fpage>207</fpage>&#x02013;<lpage>244</lpage>.</citation>
</ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>G.</given-names></name> <name><surname>Stewart</surname> <given-names>C. V.</given-names></name> <name><surname>Sofka</surname> <given-names>M.</given-names></name> <name><surname>Tsai</surname> <given-names>C.-L.</given-names></name></person-group> (<year>2007</year>). <article-title>Registration of challenging image pairs: initialization, estimation, and decision</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>29</volume>, <fpage>1973</fpage>&#x02013;<lpage>1989</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2007.1116</pub-id><pub-id pub-id-type="pmid">17848778</pub-id></citation></ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>X.</given-names></name> <name><surname>Kwitt</surname> <given-names>R.</given-names></name> <name><surname>Styner</surname> <given-names>M.</given-names></name> <name><surname>Niethammer</surname> <given-names>M.</given-names></name></person-group> (<year>2017</year>). <article-title>Quicksilver: fast predictive image registration-a deep learning approach</article-title>. <source>Neuroimage</source> <volume>158</volume>, <fpage>378</fpage>&#x02013;<lpage>396</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2017.07.008</pub-id><pub-id pub-id-type="pmid">28705497</pub-id></citation></ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>J.</given-names></name> <name><surname>Jin</surname> <given-names>K.</given-names></name> <name><surname>Gu</surname> <given-names>R.</given-names></name> <name><surname>Yan</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Sun</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Color fundus photograph registration based on feature and intensity for longitudinal evaluation of diabetic retinopathy progression</article-title>. <source>Front. Phys</source>. 10, 978392. <pub-id pub-id-type="doi">10.3389/fphy.2022.978392</pub-id></citation>
</ref>
<ref id="B48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zou</surname> <given-names>B.</given-names></name> <name><surname>He</surname> <given-names>Z.</given-names></name> <name><surname>Zhao</surname> <given-names>R.</given-names></name> <name><surname>Zhu</surname> <given-names>C.</given-names></name> <name><surname>Liao</surname> <given-names>W.</given-names></name> <name><surname>Li</surname> <given-names>S.</given-names></name></person-group> (<year>2020</year>). <article-title>Non-rigid retinal image registration using an unsupervised structure-driven regression network</article-title>. <source>Neurocomputing</source> <volume>404</volume>, <fpage>14</fpage>&#x02013;<lpage>25</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2020.04.122</pub-id></citation>
</ref>
</ref-list> 
</back>
</article> 