<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Phys.</journal-id>
<journal-title>Frontiers in Physics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Phys.</abbrev-journal-title>
<issn pub-type="epub">2296-424X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1269638</article-id>
<article-id pub-id-type="doi">10.3389/fphy.2023.1269638</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Physics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Spatio-temporal interactive fusion based visual object tracking method</article-title>
<alt-title alt-title-type="left-running-head">Huang et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fphy.2023.1269638">10.3389/fphy.2023.1269638</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Huang</surname>
<given-names>Dandan</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yu</surname>
<given-names>Siyu</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/2359721/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Duan</surname>
<given-names>Jin</given-names>
</name>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2144130/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Yingzhi</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yao</surname>
<given-names>Anni</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Yiwen</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xi</surname>
<given-names>Junhan</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
</contrib-group>
<aff>
<institution>College of Electronic Information Engineering</institution>, <institution>Changchun University of Science and Technology</institution>, <addr-line>Changchun</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/54417/overview">Gang (Gary) Ren</ext-link>, The Molecular Foundry, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2402183/overview">Fasheng Wang</ext-link>, Dalian Nationalities University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1373832/overview">Guohui Wang</ext-link>, Xi&#x2019;an Technological University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Jin Duan, <email>duanjin@vip.sina.com</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>29</day>
<month>11</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>11</volume>
<elocation-id>1269638</elocation-id>
<history>
<date date-type="received">
<day>30</day>
<month>07</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>02</day>
<month>10</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Huang, Yu, Duan, Wang, Yao, Wang and Xi.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Huang, Yu, Duan, Wang, Yao, Wang and Xi</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Visual object tracking tasks often struggle with utilizing inter-frame correlation information and handling challenges like local occlusion, deformations, and background interference. To address these issues, this paper proposes a spatio-temporal interactive fusion (STIF) based visual object tracking method. The goal is to fully utilize spatio-temporal background information, enhance feature representation for object recognition, improve tracking accuracy, adapt to object changes, and reduce model drift. The proposed method incorporates feature-enhanced networks in both temporal and spatial dimensions. It leverages spatio-temporal background information to extract salient features that contribute to improved object recognition and tracking accuracy. Additionally, the model&#x2019;s adaptability to object changes is enhanced, and model drift is minimized. A spatio-temporal interactive fusion network is employed to learn a similarity metric between the memory frame and the query frame by utilizing feature enhancement. This fusion network effectively filters out stronger feature representations through the interactive fusion of information. The proposed tracking method is evaluated on four challenging public datasets. The results demonstrate that the method achieves state-of-the-art (SOTA) performance and significantly improves tracking accuracy in complex scenarios affected by local occlusion, deformations, and background interference. Finally, the method achieves a remarkable success rate of 78.8% on TrackingNet, a large-scale tracking dataset.</p>
</abstract>
<kwd-group>
<kwd>object tracking</kwd>
<kwd>spatio-temporal context</kwd>
<kwd>feature enhancement</kwd>
<kwd>feature fusion</kwd>
<kwd>attention mechanism</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Mathematical Physics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Visual object tracking technology is one of the important research directions in the field of computer vision, which is widely used in intelligent surveillance, unmanned driving, human&#x2013;computer interaction, <italic>etc.</italic> The tracking method using correlation filtering is shown in [<xref ref-type="bibr" rid="B1">1</xref>], but with the emergence of SiamFC [<xref ref-type="bibr" rid="B2">2</xref>], the Siamese network-based tracking framework has become the mainstream of the single-object tracking algorithm framework, and a series of tracking algorithms have been generated based on it. However, such tracking algorithms have the following shortcomings:</p>
<p>1) Most of the current Siamese network-based trackers generally use the initial frame as a template tracking strategy, which makes it difficult for the algorithm to adapt to situations such as severe object deformation and occlusion, resulting in poor robustness and accuracy. In addition, these trackers use only the appearance information of the current frame, failing to take full advantage of the abundant temporal contextual information in the historical frame sequences and ignoring the temporal and spatial correlation of the objects.</p>
<p>To improve this situation, some trackers introduce a template update mechanism or use multiple templates, such as UpdateNet [<xref ref-type="bibr" rid="B3">3</xref>], while others consider the spatio-temporal correlation and introduce an attention mechanism approach for improvement, such as SiamAttn [<xref ref-type="bibr" rid="B4">4</xref>] and SiamADT [<xref ref-type="bibr" rid="B5">5</xref>]. The aforementioned approaches can enhance the robustness of the tracker to some extent, but such strategies are mainly of limited use and inevitably increase the computational effort.</p>
<p>2) Most of the existing popular template frame feature and query frame feature trackers still use correlation operations for fusion, such as SiamFC, which causes the lack of semantic information and global information to some extent. The absence of all this information causes such tracking algorithms to encounter difficulty adapting to changes in the appearance of the object in the face of challenges, such as local occlusions and deformations, thus reducing the tracking performance of the algorithms and considerably limiting their use.</p>
<p>In response to the aforementioned analysis, this paper stores multiple historical frame information as memory frames in the tracking process and enhances the features in both temporal and spatial dimensions, with the aim of breaking the conventional pattern that most algorithms invariably use the initial frame <italic>a priori</italic> information as the tracking template and completely exploiting the hidden spatio-temporal contextual information in the historical frame sequence, while enabling the tracker to better adapt to the changing appearance of the object; in order to improve tracking accuracy to achieve a strict comparison between candidate objects and template, this paper designs a spatio-temporal interactive fusion (STIF) network for establishing the relationship between memory frame and query frame features and also obtains a more robust global feature representation by feature interaction between the two based on feature enhancement. By mining and exploiting the aforementioned information, the proposed algorithm improves the accuracy and robustness of the tracking model. The main contributions of this paper are summarized as follows.<list list-type="simple">
<list-item>
<p>&#x2022; An end-to-end spatio-temporal interactive fusion object tracking framework is proposed. The whole network is not only simple in structure but also has a strong adaptive capability to the changing appearance of the object in different scenarios.</p>
</list-item>
<list-item>
<p>&#x2022; A feature enhancement network is established through which the feature sequences are processed to capture the dynamic features of the object in both the temporal and spatial dimensions. The network can achieve a meticulous capture of the object information in all directions, making the extracted object features more significant, thus leading to higher accuracy of the tracking algorithm.</p>
</list-item>
<list-item>
<p>&#x2022; A spatio-temporal interactive fusion network is proposed, based on the feature enhancement network, to achieve fine alignment and matching of the features of the two branches through the mutual transfer and influence of information between the memory and query branches and enhance the feature expression capability, which can effectively solve the problem of model drift caused by occlusion and deformation<italic>.</italic> The adversarial interference capability of the model will be improved during the tracking process to achieve more robust tracking.</p>
</list-item>
</list>
</p>
</sec>
<sec id="s2">
<title>2 Related work</title>
<p>In recent years, the rapid development of object tracking technology has led to the emergence of many object tracking algorithms. Among them, the emergence of Siamese network-based object tracking methods has transformed the search problem into similarity matching and solved the previous problem of large computational effort for visual tracking tasks; however, in the reasoning process of such algorithms, the given template information is generally intercepted from the first frame of the video sequence and matched with the current search region for feature information to achieve tracking. It has excellent performance and real-time tracking speed in many traditional tracking scenarios. Facing the aforementioned problems, DSiam [<xref ref-type="bibr" rid="B7">7</xref>] proposed a tracking algorithm based on the dynamic Siamese network, which uses a dynamic model to analyze the motion of the object and improves the accuracy and stability of tracking by dynamic feature extraction and template update. In addition, UpdateNet linearly updated the object template using a moving average method with a fixed learning rate and proposed an adaptive template updating strategy that combines the initial frame template, the accumulated template, and the current frame template to achieve the prediction of the best template for the next frame. STARK [<xref ref-type="bibr" rid="B8">8</xref>] sets up a dynamic template and determines whether the dynamic template is updated by setting a threshold on the prediction confidence. STMTrack [<xref ref-type="bibr" rid="B9">9</xref>] proposes a tracking framework based on spatio-temporal memory networks, which can make full use of the historical information associated with the object, thus avoiding template updates and achieving a template-free framework.</p>
<p>In addition, when the tracking object contains multiple spatial dimensions as well as a temporal dimension, the spatio-temporal information needs to be used to focus on the dynamic characteristics of the object at different locations and different moments for more accurate tracking. In order to make full use of spatio-temporal contextual information, object tracking methods based on attention mechanisms are also widely used. [<xref ref-type="bibr" rid="B10">10</xref>] introduced an encoder&#x2013;decoder attention module to filter different features by compressing the feature map and establishing relationships between channels in the Siamese network. Efficient visual tracking with a stacked channel&#x2013;spatial attention (SCSAtt) mechanism [<xref ref-type="bibr" rid="B11">11</xref>] improves the accuracy of the model by introducing an attention mechanism to explore the object features and designing a linearly stacked attention mechanism. SA-Siam [<xref ref-type="bibr" rid="B12">12</xref>] extracts different features of the object through a dual-branching structure of semantics and appearance and uses a channel attention mechanism for feature selection of the object but ignores template updates.</p>
<p>Different from these methods, our network considers spatio-temporal inter-frame correlation and makes full use of multiple historical frames of the tracking process, enhances features, and improves focus on candidate object areas by capturing dynamic features of the object in time and space. Based on this, the spatio-temporal interaction fusion network communicates the feature information of the memory branch with the query branch to achieve secondary screening of features and further enhances the feature expression capability to obtain global features, which can significantly improve the tracking performance of the model in challenging scenarios.</p>
</sec>
<sec sec-type="methods" id="s3">
<title>3 Methods</title>
<p>In this section, we first introduce the general framework of the proposed spatio-temporal interactive fusion-based object tracking method. In addition, we describe each part in detail in <xref ref-type="sec" rid="s3-1">Section 3.1</xref>, <xref ref-type="sec" rid="s3-2">Section 3.2</xref>, and <xref ref-type="sec" rid="s3-3">Section 3.3</xref>.</p>
<sec id="s3-1">
<title>3.1 Feature enhancement network</title>
<p>The spatio-temporal association between frames can enrich the template information, while the use of the spatio-temporal context can accurately capture the temporal and spatial changes of the object, ensuring that the features have sufficient characterization power. We use spatio-temporal feature enhancement to enrich the template information and query the features of the frames. The feature enhancement module is mainly divided into two parts: memory frame enhancement and query frame enhancement. The memory frame enhancement includes both temporal and spatial dimensions to enhance the temporality and expressiveness of the features, while for the query frame, as it only provides the image information of the current moment, only spatial enhancement is used, focusing more on the expression and extraction of spatial information. <xref ref-type="fig" rid="F1">Figure 1</xref> shows the overall framework of our network.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Schematic diagram of the processing flow of the algorithm in this paper. The backbone network is used to extract the memory frame features and query frame features as inputs to the overall structure. Subsequently, the feature enhancement operation is carried out, where the memory frame performs the spatio-temporal dimension enhancement (STE and SDE) operation to fully combine the spatio-temporal context information to enrich the template information, and the query frame performs the spatial dimension enhancement (SDE) operation to improve the focus of the object region. Then, spatio-temporal interactive fusion (STIF) of features is carried out to achieve the interaction of the two branches of features, and the final feature map is obtained and sent to the head network for tracking result estimation. Reproduced from the OTB100 dataset, [<xref ref-type="bibr" rid="B20">20</xref>], with permission from IEEE.</p>
</caption>
<graphic xlink:href="fphy-11-1269638-g001.tif"/>
</fig>
<sec id="s3-1-1">
<title>3.1.1 Memory frame enhancement</title>
<p>The memory frame features are mainly used in a spatio-temporal enhancement (STE) network, as shown in <xref ref-type="fig" rid="F2">Figure 2</xref>, which mainly includes time dimension enhancement (TDE) and spatial dimension enhancement (SDE). The features are adaptively optimized by assigning different weights to each time and location based on the response to the object. End-to-end training can be achieved without adding additional parameters using spatio-temporal feature enhancement networks. This process is shown in the following equation:<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mi mathvariant="bold-italic">T</mml:mi>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">N</mml:mi>
<mml:mi mathvariant="bold-italic">T</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mi mathvariant="bold-italic">m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2297;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mi mathvariant="bold-italic">m</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mi mathvariant="bold-italic">S</mml:mi>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">N</mml:mi>
<mml:mi mathvariant="bold-italic">S</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mi mathvariant="bold-italic">T</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2297;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mi mathvariant="bold-italic">T</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <inline-formula id="inf1">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the memory branch input feature, <inline-formula id="inf2">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>T</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the feature processed by STE, <inline-formula id="inf3">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the feature processed by SDE, <inline-formula id="inf4">
<mml:math id="m5">
<mml:mrow>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf5">
<mml:math id="m6">
<mml:mrow>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>S</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> are the features enhanced by STE and SDE, respectively, and <inline-formula id="inf6">
<mml:math id="m7">
<mml:mrow>
<mml:mo>&#x2297;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> denotes element-wise multiplication.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Spatio-temporal feature enhancement network structure.</p>
</caption>
<graphic xlink:href="fphy-11-1269638-g002.tif"/>
</fig>
<p>Due to the temporal correlation between video sequence frames, multiple consecutive frames are used as reference frames to capture subtle appearance variations in the object during motion, thus enhancing the discriminative features in the tracking instances. Therefore, in STE, we mainly use the T memory frames adjacent to the frame to be tracked as reference frames, and the overall process is as follows: first, we use the feature extraction network to obtain the memory frame feature map <inline-formula id="inf7">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>; then, we feed the feature map <inline-formula id="inf8">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> into the STE network; finally, we obtain the enhanced features <inline-formula id="inf9">
<mml:math id="m10">
<mml:mrow>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> after the <inline-formula id="inf10">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> time-attentive network according to the following strategy:<disp-formula id="e2">
<mml:math id="m12">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mi mathvariant="bold-italic">T</mml:mi>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="bold-italic">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c9;</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:msub>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
</mml:msub>
</mml:msub>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf11">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c9;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the fusion weight of the corresponding reference frame, and the reference frame weight is positively correlated with its contribution to the tracking template. <inline-formula id="inf12">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the feature of the <italic>i</italic>th reference frame. Since the initial frame contains more comprehensive information about the appearance of the object, it has a certain reference value when the object is occluded or motion blur occurs. The similarity between the reference frame features and the initial frame features is computed to obtain the fusion weight coefficients:<disp-formula id="e3">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c9;</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="bold-italic">S</mml:mi>
<mml:mi mathvariant="bold-italic">o</mml:mi>
<mml:mi mathvariant="bold-italic">f</mml:mi>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mi mathvariant="bold-italic">M</mml:mi>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:msub>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
</mml:msub>
</mml:msub>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:msub>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mn mathvariant="bold">0</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">T</mml:mi>
<mml:mi mathvariant="bold-italic">r</mml:mi>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:mi mathvariant="bold-italic">n</mml:mi>
<mml:mi mathvariant="bold-italic">s</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:msqrt>
<mml:mi mathvariant="bold-italic">C</mml:mi>
</mml:msqrt>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where <inline-formula id="inf13">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the video initial frame image feature, C is the number of channels, and <inline-formula id="inf14">
<mml:math id="m17">
<mml:mrow>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the transpose of the initial frame. Finally, these weights are applied to the feature weights, and the adaptive assignment of fusion weights can be implemented according to Eqs <xref ref-type="disp-formula" rid="e2">2</xref>, <xref ref-type="disp-formula" rid="e3">3</xref> to obtain the temporally enhanced memory frame feature output.</p>
<p>Unlike temporal dimension enhancement, spatial dimension enhancement focuses on the location information of the object in the feature image and improves the sensitivity of the network to spatial feature information by highlighting or weakening the feature information at different spatial locations. The specific procedure is shown in the following equation:<disp-formula id="e4">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3be;</mml:mi>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold-italic">f</mml:mi>
<mml:mrow>
<mml:mn mathvariant="bold">7</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn mathvariant="bold">7</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mi mathvariant="bold-italic">C</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">c</mml:mi>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:mi mathvariant="bold-italic">t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:mi mathvariant="bold-italic">v</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
<mml:mi mathvariant="bold-italic">S</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mo>_</mml:mo>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
<mml:mi mathvariant="bold-italic">S</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
</p>
<p>where <inline-formula id="inf15">
<mml:math id="m19">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3be;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the feature output after the aforementioned processing; <inline-formula id="inf16">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#x2219;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mo>&#x2219;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the connected maximum pooling and average pooling; <inline-formula id="inf17">
<mml:math id="m21">
<mml:mrow>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mn>7</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the convolution layer with the convolution kernel of 7 &#xd7; 7; <inline-formula id="inf18">
<mml:math id="m22">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the sigmoid function normalization; and <inline-formula id="inf19">
<mml:math id="m23">
<mml:mrow>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>_</mml:mo>
<mml:mi>a</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mi>S</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf20">
<mml:math id="m24">
<mml:mrow>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>_</mml:mo>
<mml:mi>a</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> are the features of <inline-formula id="inf21">
<mml:math id="m25">
<mml:mrow>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> after global average pooling and maximum pooling, respectively. The final spatially enhanced memory frame features are shown as<disp-formula id="e5">
<mml:math id="m26">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mi mathvariant="bold-italic">S</mml:mi>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3be;</mml:mi>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:msub>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mi mathvariant="bold-italic">T</mml:mi>
</mml:msubsup>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
</sec>
<sec id="s3-1-2">
<title>3.1.2 Query frame enhancement</title>
<p>For the query frame, as it only provides information about the image at the current moment, only spatial enhancement is used, focusing more on the representation and extraction of spatial information. Spatial enhancement is mainly used to improve the spatial sensitivity and accuracy of the features by optimizing them, making them more accurate in reflecting spatial information such as the morphology and edges of the object.</p>
<p>Like memory frame enhancement, the result of processing after spatial pooling is shown in Eq. <xref ref-type="disp-formula" rid="e6">6</xref>:<disp-formula id="e6">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3be;</mml:mi>
<mml:mn mathvariant="bold">2</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold-italic">f</mml:mi>
<mml:mrow>
<mml:mn mathvariant="bold">7</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn mathvariant="bold">7</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mi mathvariant="bold-italic">C</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">c</mml:mi>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:mi mathvariant="bold-italic">t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:mi mathvariant="bold-italic">v</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
<mml:mi mathvariant="bold-italic">S</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mo>_</mml:mo>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
<mml:mi mathvariant="bold-italic">S</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>where <inline-formula id="inf22">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3be;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the feature output in the query branch after the aforementioned processing. The final spatially enhanced query frame features are shown as<disp-formula id="e7">
<mml:math id="m29">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mi mathvariant="bold-italic">q</mml:mi>
<mml:mi mathvariant="bold-italic">S</mml:mi>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3be;</mml:mi>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:msub>
<mml:msub>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mi mathvariant="bold-italic">q</mml:mi>
</mml:msub>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
</p>
</sec>
</sec>
<sec id="s3-2">
<title>3.2 Spatio-temporal interactive fusion network</title>
<p>The main purpose of the spatio-temporal interactive fusion (STIF) network is to compare and complement features from different frames to achieve better feature representation and tracking results. As shown in <xref ref-type="fig" rid="F3">Figure 3</xref>, the memory branch and the query branch after feature enhancement are used as network inputs. By cross-comparing the features of the memory and query frames and emphasizing the importance and priority between different features, the dynamic features and correlations of the object in time and space can be effectively captured, and the features are fused into more robust and accurate features to achieve effective information interaction and fusion between the two branches. The attention model used in the network is defined as shown in Eq. <xref ref-type="disp-formula" rid="e8">8</xref>, with three inputs, Q, K, and V, as query, key, and value, respectively, and <inline-formula id="inf23">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the dimensionality of the key.<disp-formula id="e8">
<mml:math id="m31">
<mml:mrow>
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mi mathvariant="bold-italic">e</mml:mi>
<mml:mi mathvariant="bold-italic">n</mml:mi>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mi mathvariant="bold-italic">o</mml:mi>
<mml:mi mathvariant="bold-italic">n</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">V</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="bold-italic">S</mml:mi>
<mml:mi mathvariant="bold-italic">o</mml:mi>
<mml:mi mathvariant="bold-italic">f</mml:mi>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mi mathvariant="bold-italic">M</mml:mi>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
<mml:msup>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mi mathvariant="bold-italic">T</mml:mi>
</mml:msup>
</mml:mrow>
<mml:msqrt>
<mml:msub>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msub>
</mml:msqrt>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>spatio-temporal interactive fusion network structure.</p>
</caption>
<graphic xlink:href="fphy-11-1269638-g003.tif"/>
</fig>
<p>Specifically, based on the feature enhancement network processing, we use the memory frame feature <inline-formula id="inf24">
<mml:math id="m32">
<mml:mrow>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>S</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and the query frame feature <inline-formula id="inf25">
<mml:math id="m33">
<mml:mrow>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mi>q</mml:mi>
<mml:mi>S</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> obtained by the enhancement process as inputs to the spatio-temporal interactive fusion network. Taking the memory branch as an example, first, consider <inline-formula id="inf26">
<mml:math id="m34">
<mml:mrow>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>S</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> as V and K of the memory branch and <inline-formula id="inf27">
<mml:math id="m35">
<mml:mrow>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mi>q</mml:mi>
<mml:mi>S</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> as Q of the memory branch; the similarity measure between the memory frame features and the query frame features can be obtained by computing the similarity between K and Q, and the corresponding similarity weights are then weighted with the memory frame features to enable feature information interaction between the memory branch and the query branch. In addition, before feature fusion, we use a gating unit to adjust the features, mainly to strengthen the nonlinear nature and generalization ability of the fusion network to avoid gradient explosion and increase the number of parameters during model training. Finally, the memory frame features after the interaction are combined with the query frame features in the second dimension to obtain the final fused feature map y. The query branch operation is the same as mentioned previously. The specific process is formulated as follows:<disp-formula id="e9">
<mml:math id="m36">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">y</mml:mi>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">C</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">c</mml:mi>
<mml:mi mathvariant="bold-italic">o</mml:mi>
<mml:mi mathvariant="bold-italic">n</mml:mi>
<mml:mi mathvariant="bold-italic">v</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mi mathvariant="bold-italic">n</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mi mathvariant="bold-italic">S</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mi mathvariant="bold-italic">q</mml:mi>
<mml:mi mathvariant="bold-italic">S</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
<disp-formula id="e10">
<mml:math id="m37">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">y</mml:mi>
<mml:mn mathvariant="bold">2</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">C</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">c</mml:mi>
<mml:mi mathvariant="bold-italic">o</mml:mi>
<mml:mi mathvariant="bold-italic">n</mml:mi>
<mml:mi mathvariant="bold-italic">v</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mi mathvariant="bold-italic">n</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mi mathvariant="bold-italic">q</mml:mi>
<mml:mi mathvariant="bold-italic">S</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mi mathvariant="bold-italic">S</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>
<disp-formula id="e11">
<mml:math id="m38">
<mml:mrow>
<mml:mi mathvariant="bold-italic">y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="bold-italic">c</mml:mi>
<mml:mi mathvariant="bold-italic">o</mml:mi>
<mml:mi mathvariant="bold-italic">n</mml:mi>
<mml:mi mathvariant="bold-italic">c</mml:mi>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">y</mml:mi>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">y</mml:mi>
<mml:mn mathvariant="bold">2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>where <inline-formula id="inf28">
<mml:math id="m39">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the memory frame features after interaction, <inline-formula id="inf29">
<mml:math id="m40">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the query frame features after interaction, <inline-formula id="inf30">
<mml:math id="m41">
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#x2219;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mo>&#x2219;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the fusion join operation; <inline-formula id="inf31">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes a convolutional layer with a convolutional kernel of 3 <inline-formula id="inf32">
<mml:math id="m43">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 3, and Attn denotes the attention operation.</p>
</sec>
<sec id="s3-3">
<title>3.3 Classification regression head network</title>
<p>Inspired by the fact that the first-level anchorless detector [<xref ref-type="bibr" rid="B13">13</xref>] has better detection and fewer computational parameters than the first-level detector [<xref ref-type="bibr" rid="B14">14</xref>] based on anchor frame point boxes for object detection, we employ an anchorless head network that contains a classification branch to classify the information in the image and an anchorless regression branch for direct estimation of the object&#x2019;s bounding box.</p>
<p>A lightweight classification convolutional network <inline-formula id="inf33">
<mml:math id="m44">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c9;</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is first used to encode the fused feature mapping, which combines information from memory frames and query frames and is more suitable for the classification task. The output dimension of <inline-formula id="inf34">
<mml:math id="m45">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c9;</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is then reduced to 1 using a linear convolutional layer with a 1 &#xd7; 1 kernel, and finally, the classification response map <inline-formula id="inf35">
<mml:math id="m46">
<mml:mrow>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is obtained.</p>
<p>Furthermore, since positive samples close to the object boundary tend to predict low-quality object bounding boxes, a branch is bifurcated after <inline-formula id="inf36">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c9;</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> for the centrality response map <inline-formula id="inf37">
<mml:math id="m48">
<mml:mrow>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, as shown on the right side of <xref ref-type="fig" rid="F1">Figure 1</xref>. In the inference process, <inline-formula id="inf38">
<mml:math id="m49">
<mml:mrow>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is multiplied with <inline-formula id="inf39">
<mml:math id="m50">
<mml:mrow>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> to suppress the classification confidence score of pixels at a distance from the object center. In the regression branch, we use the feature mapping y for another lightweight regression convolutional network <inline-formula id="inf40">
<mml:math id="m51">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c9;</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and then reduce the dimensionality of the output features to 4 to generate the regression response mapping <inline-formula id="inf41">
<mml:math id="m52">
<mml:mrow>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mn>4</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. Finally, the classification loss <inline-formula id="inf42">
<mml:math id="m53">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and regression loss <inline-formula id="inf43">
<mml:math id="m54">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are used as functions, as described in [<xref ref-type="bibr" rid="B7">7</xref>, <xref ref-type="bibr" rid="B15">15</xref>], respectively, and the final overall loss function can be expressed as<disp-formula id="e12">
<mml:math id="m55">
<mml:mrow>
<mml:mi mathvariant="bold-italic">L</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">L</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">c</mml:mi>
<mml:mi mathvariant="bold-italic">l</mml:mi>
<mml:mi mathvariant="bold-italic">s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3bb;</mml:mi>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:msub>
<mml:msub>
<mml:mi mathvariant="bold-italic">L</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">r</mml:mi>
<mml:mi mathvariant="bold-italic">e</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3bb;</mml:mi>
<mml:mn mathvariant="bold">2</mml:mn>
</mml:msub>
<mml:msub>
<mml:mi mathvariant="bold-italic">L</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">c</mml:mi>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mi mathvariant="bold-italic">r</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>where <inline-formula id="inf44">
<mml:math id="m56">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf45">
<mml:math id="m57">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are both weighting factors.</p>
</sec>
</sec>
<sec id="s4">
<title>4 Experimental results and analysis</title>
<sec id="s4-1">
<title>4.1 Experimental details</title>
<sec id="s4-1-1">
<title>4.1.1 Datasets</title>
<p>In the training phase, we mainly used the training sets of TrackingNet [<xref ref-type="bibr" rid="B15">15</xref>], LaSOT [<xref ref-type="bibr" rid="B16">16</xref>], GOT-10k [<xref ref-type="bibr" rid="B17">17</xref>], ILSVRC VID [<xref ref-type="bibr" rid="B18">18</xref>], ILSVRC DET [<xref ref-type="bibr" rid="B18">18</xref>], and COCO [<xref ref-type="bibr" rid="B19">19</xref>] as the training datasets except for the GOT-10k [<xref ref-type="bibr" rid="B18">18</xref>] benchmark; in the testing phase, we mainly used TrackingNet [<xref ref-type="bibr" rid="B15">15</xref>], LaSOT [<xref ref-type="bibr" rid="B16">16</xref>], GOT-10k [<xref ref-type="bibr" rid="B17">17</xref>], OTB100 [<xref ref-type="bibr" rid="B20">20</xref>], and WATB [<xref ref-type="bibr" rid="B21">21</xref>]datasets for testing and comparing with other object tracking methods. In addition, four sequences with occlusion and appearance deformation properties were extracted from the LaSOT [<xref ref-type="bibr" rid="B16">16</xref>] dataset for testing.</p>
</sec>
<sec id="s4-1-2">
<title>4.1.2 Experimental setting</title>
<sec id="s4-1-2-1">
<title>4.1.2.1 Model setup</title>
<p>We used the PyTorch [<xref ref-type="bibr" rid="B22">22</xref>] framework for our experiments, employing GoogLeNet [<xref ref-type="bibr" rid="B23">23</xref>] as the backbone of our feature extraction network <inline-formula id="inf46">
<mml:math id="m58">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c6;</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>&#x3c6;</mml:mi>
<mml:mi>q</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and both the classification convolutional network <inline-formula id="inf47">
<mml:math id="m59">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c9;</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the regression convolutional network <inline-formula id="inf48">
<mml:math id="m60">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c9;</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> consisting of seven convolutional layers. Each convolutional layer in <inline-formula id="inf49">
<mml:math id="m61">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c9;</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf50">
<mml:math id="m62">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c9;</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is followed by a ReLU activation function.</p>
</sec>
<sec id="s4-1-2-2">
<title>4.1.2.2 Optimization and training strategies</title>
<p>We mainly use SDG to optimize the loss function during the training process. A total of 20 epochs are set for the experiment, and for the GOT-10k benchmark test, the number of samples per epoch is set to 150,000, the mini-batch size is set to 20, the momentum decay rate is 0.9, the weight decay rate is <inline-formula id="inf51">
<mml:math id="m63">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mn>10</mml:mn>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, the initial learning rate is set to <inline-formula id="inf52">
<mml:math id="m64">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mn>10</mml:mn>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf53">
<mml:math id="m65">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mn>10</mml:mn>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>8</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> in the first epoch when the backbone network is trained, and in the other epochs, the learning rate increases from <inline-formula id="inf54">
<mml:math id="m66">
<mml:mrow>
<mml:mn>8</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mn>10</mml:mn>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf55">
<mml:math id="m67">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mn>10</mml:mn>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>6</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. In the online tracking process, the input memory frame template image size is <inline-formula id="inf56">
<mml:math id="m68">
<mml:mrow>
<mml:mn>289</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>289</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, the search image is <inline-formula id="inf57">
<mml:math id="m69">
<mml:mrow>
<mml:mn>289</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>289</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, where the memory frame is stored in T &#x3d; 3, and the image pairs are entered into the respective network branches to finally obtain a score map of size 25. Furthermore, in the hyperparameter settings, &#x3bb;1 and &#x3bb;2 are the two hyperparameters that control the weights between the three losses, and we set them to the default value 1.</p>
</sec>
</sec>
</sec>
<sec id="s4-2">
<title>4.2 Ablation experiments</title>
<p>To verify the effectiveness of the proposed spatio-temporal interactive fusion-based object tracking method using the spatio-temporal feature enhancement module in the memory branch, only spatial dimension enhancement (SDE) in the query branch, and the spatio-temporal interaction model (STIF), we performed a series of ablation experiments on the GOT-10k [<xref ref-type="bibr" rid="B17">17</xref>] dataset, and it should be noted that the baseline tracker we compared were mainly STMTrack tracker. The specific experiments are as follows: 1) adding STE, SDE, and STIF separately to the overall network framework to compare with the baseline network; 2) adding STE and SDE simultaneously to the network framework (i.e., using the feature enhancement network as a whole) to compare with STE and SDE alone; 3) adding STE, SDE, and STIF simultaneously to the network framework to compare with the training performance when each module is added separately.</p>
<p>The results of the ablation experiments are shown in <xref ref-type="table" rid="T1">Table 1</xref>, where AO represents the average overlap rate and SR represents the success rate. From <xref ref-type="table" rid="T1">Table 1</xref>, for AO, adding each network individually improves the value. While the overall usage of the feature enhancement network improves by approximately 2% compared to adding STE and SDE networks alone, adding STE, SDE, and STIF simultaneously to the network framework achieves the best performance, improving the value of AO by 10% compared to the baseline network. In addition, the results in <xref ref-type="table" rid="T1">Table 1</xref> show that the STE, SDE, and STIF modules achieve more significant improvements in the tracking success rate metrics SR0.5 and SR0.75. The results of the ablation experiments sufficiently demonstrate the effectiveness of the feature enhancement module and the interaction fusion module designed in this paper.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Comparison of the tracking performances of different networks in the same experimental environment and datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Dataset</th>
<th align="center">Baseline tracker</th>
<th align="center">Fm &#x2b; STE</th>
<th align="center">Fq &#x2b; SDE</th>
<th align="center">&#x2b;STIF</th>
<th align="center">AO</th>
<th align="center">
<inline-formula id="inf58">
<mml:math id="m70">
<mml:mrow>
<mml:msub>
<mml:mtext>SR</mml:mtext>
<mml:mn mathvariant="bold">0.5</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th align="center">
<inline-formula id="inf59">
<mml:math id="m71">
<mml:mrow>
<mml:msub>
<mml:mtext>SR</mml:mtext>
<mml:mn mathvariant="bold">0.75</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">GOT-10k</td>
<td align="center">
<inline-formula id="inf60">
<mml:math id="m72">
<mml:mrow>
<mml:mo>&#x221a;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center">0.642</td>
<td align="center">0.737</td>
<td align="center">0.545</td>
</tr>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center">
<inline-formula id="inf61">
<mml:math id="m73">
<mml:mrow>
<mml:mo>&#x221a;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center"/>
<td align="center"/>
<td align="center">0.644</td>
<td align="center">0.739</td>
<td align="center">0.550</td>
</tr>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center">
<inline-formula id="inf62">
<mml:math id="m74">
<mml:mrow>
<mml:mo>&#x221a;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center"/>
<td align="center">0.643</td>
<td align="center">0.740</td>
<td align="center">0.550</td>
</tr>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center">
<inline-formula id="inf63">
<mml:math id="m75">
<mml:mrow>
<mml:mo>&#x221a;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">0.644</td>
<td align="center">0.742</td>
<td align="center">0.549</td>
</tr>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center">
<inline-formula id="inf64">
<mml:math id="m76">
<mml:mrow>
<mml:mo>&#x221a;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">
<inline-formula id="inf65">
<mml:math id="m77">
<mml:mrow>
<mml:mo>&#x221a;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center"/>
<td align="center">0.647</td>
<td align="center">0.750</td>
<td align="center">0.575</td>
</tr>
<tr>
<td align="center"/>
<td align="center"/>
<td align="center">
<inline-formula id="inf66">
<mml:math id="m78">
<mml:mrow>
<mml:mo>&#x221a;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">
<inline-formula id="inf67">
<mml:math id="m79">
<mml:mrow>
<mml:mo>&#x221a;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">
<inline-formula id="inf68">
<mml:math id="m80">
<mml:mrow>
<mml:mo>&#x221a;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">0.652</td>
<td align="center">0.781</td>
<td align="center">0.590</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4-3">
<title>4.3 Comparison experiments</title>
<p>In order to evaluate the performance of our method, this section compares the performance of the proposed tracking algorithm with the current mainstream algorithms on the datasets of GOT-10k [<xref ref-type="bibr" rid="B17">17</xref>], LaSOT [<xref ref-type="bibr" rid="B16">16</xref>], TrackingNet [<xref ref-type="bibr" rid="B15">15</xref>], OTB100 [<xref ref-type="bibr" rid="B20">20</xref>],and WATB [<xref ref-type="bibr" rid="B21">21</xref>].</p>
<sec id="s4-3-1">
<title>4.3.1 Evaluation on the GOT-10k dataset</title>
<p>To evaluate the generalization ability of our method, we choose the GOT-10k dataset, which is a large-scale generic benchmark dataset with more than 10,000 videos of real-life scenarios and a test set with 180 video sequences. A key feature of GOT-10k is that there is no overlap between the classes of tracked objects in the training and test sets, which can be used to evaluate the generalization ability of the tracker. To ensure a fair comparison, this paper follows the GOT-10k testing protocol, and only the GOT-10k training set is used to train the tracker.</p>
<p>As shown in <xref ref-type="table" rid="T2">Table 2</xref>, by comparing with some existing advanced trackers, including UAST [<xref ref-type="bibr" rid="B24">24</xref>], STMTrack [<xref ref-type="bibr" rid="B9">9</xref>], and SiamGAT [<xref ref-type="bibr" rid="B25">25</xref>], the algorithm proposed in this paper has the best results, where AO is 65.2%, <inline-formula id="inf69">
<mml:math id="m81">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mn>0.5</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is 78.1%, and <inline-formula id="inf70">
<mml:math id="m82">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mn>0.75</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is 59.0%. The method in this paper mainly makes full use of the spatio-temporal context information and can improve the generalization ability of the model, thus outperforming these advanced trackers in terms of performance.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Test results of the GOT-10k dataset, where AO denotes the average overlap rate, SR denotes the success rate, and trackers are sorted by AO values from top to bottom.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Tracker</th>
<th align="center">AO</th>
<th align="center">
<inline-formula id="inf71">
<mml:math id="m83">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">S</mml:mi>
<mml:mi mathvariant="bold-italic">R</mml:mi>
</mml:mrow>
<mml:mn mathvariant="bold">0.5</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th align="center">
<inline-formula id="inf72">
<mml:math id="m84">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">S</mml:mi>
<mml:mi mathvariant="bold-italic">R</mml:mi>
</mml:mrow>
<mml:mn mathvariant="bold">0.75</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">SiamFC [<xref ref-type="bibr" rid="B2">2</xref>]</td>
<td align="center">0.348</td>
<td align="center">0.353</td>
<td align="center">0.098</td>
</tr>
<tr>
<td align="center">SiamRPN&#x2b;&#x2b; [<xref ref-type="bibr" rid="B26">26</xref>]</td>
<td align="center">0.517</td>
<td align="center">0.616</td>
<td align="center">0.325</td>
</tr>
<tr>
<td align="center">DiMP [<xref ref-type="bibr" rid="B27">27</xref>]</td>
<td align="center">0.611</td>
<td align="center">0.717</td>
<td align="center">0.492</td>
</tr>
<tr>
<td align="center">Ocean [<xref ref-type="bibr" rid="B28">28</xref>]</td>
<td align="center">0.611</td>
<td align="center">0.721</td>
<td align="center">0.473</td>
</tr>
<tr>
<td align="center">SiamGAT [<xref ref-type="bibr" rid="B6">6</xref>]</td>
<td align="center">0.627</td>
<td align="center">0.743</td>
<td align="center">0.575</td>
</tr>
<tr>
<td align="center">STMTrack [<xref ref-type="bibr" rid="B9">9</xref>]</td>
<td align="center">0.642</td>
<td align="center">0.737</td>
<td align="center">0.575</td>
</tr>
<tr>
<td align="center">UAST [<xref ref-type="bibr" rid="B24">24</xref>]</td>
<td align="center">0.648</td>
<td align="center">0.751</td>
<td align="center">0.578</td>
</tr>
<tr>
<td align="center">
<bold>Ours</bold>
</td>
<td align="center">
<bold>0.652</bold>
</td>
<td align="center">
<bold>0.781</bold>
</td>
<td align="center">
<bold>0.590</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The meaning of the bold values in table primarily represents the training results of the methods proposed in this article.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s4-3-2">
<title>4.3.2 Evaluation on the LaSOT dataset</title>
<p>To evaluate the adaptability of our method in the face of different scenario variations, we tested it using the LaSOT dataset. LaSOT is also a large-scale single-object tracking dataset with high-quality annotations. Its test set consists of 280 long videos, with an average of 2,500 frames each, including 70 tracking object classes, each containing 20 tracking sequences, containing many video sequences with different attributes, covering a wide range of scenarios at multiple scales, speeds, backgrounds, and poses. LaSOT analyzes the performance of each algorithm mainly using accuracy maps based on position error metrics and success maps based on overlap metrics.</p>
<p>The success plot is shown in <xref ref-type="fig" rid="F4">Figure 4</xref>, and the accuracy plot is shown in <xref ref-type="fig" rid="F5">Figure 5</xref>. Compared with a variety of comparable trackers, our tracker reaches the forefront in terms of success, accuracy, and standardized precision, including 6.7% and 6.8% improvement in the success rate and 6.8% and 7.6% improvement in accuracy compared to PACNet [<xref ref-type="bibr" rid="B29">29</xref>] and UAST [<xref ref-type="bibr" rid="B24">24</xref>], respectively.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Success rate of our method on the LaSOT dataset.</p>
</caption>
<graphic xlink:href="fphy-11-1269638-g004.tif"/>
</fig>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Precision of our method on the LaSOT dataset.</p>
</caption>
<graphic xlink:href="fphy-11-1269638-g005.tif"/>
</fig>
<p>In addition, for some complex scenario challenges, such as the similar appearance as the object (BC), the object is deformable during tracking (DEF); the object rotates in the image (ROT), and the object is partially occluded in the sequence (POC). We also report results on the LaSOT [<xref ref-type="bibr" rid="B16">16</xref>] dataset. The results shown in <xref ref-type="fig" rid="F6">Figure 6</xref> demonstrate that our method exhibits optimal performance when facing the aforementioned challenges. This indicates that our model is capable of effectively adapting to variations in the images without easily experiencing drifting, showcasing strong robustness.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Some of the challenges tested on the LaSOT dataset include partial occlusion, similar appearance, deformation, and rotation. <bold>(A, B, E, F)</bold> Results of accuracy evaluation; <bold>(C, D, G, H)</bold> results of success rate evaluation.</p>
</caption>
<graphic xlink:href="fphy-11-1269638-g006.tif"/>
</fig>
</sec>
<sec id="s4-3-3">
<title>4.3.3 Evaluation on the TrackingNet dataset</title>
<p>TrackingNet is a large-scale short-term tracking dataset that provides many field videos for training and testing. The test set contains 511 video sequences. Evaluation on the TrackingNet dataset enables testing of the model&#x2019;s tracking performance in a variety of scenarios, including the effectiveness of tracking different objects, adaptability to different locations, scales, lighting, and background changes, and evaluation of various metrics, such as the model&#x2019;s recognition and tracking accuracy for many different attributes.</p>
<p>We evaluate the tracker on the test set and obtain the results from a dedicated evaluation server. As shown in <xref ref-type="table" rid="T3">Table 3</xref>, our tracker outperforms some other advanced tracking algorithms to a large extent exhibiting better performance.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>TrackingNet dataset test results, with trackers ranked from top to bottom according to &#x201c;success&#x201d; values, where &#x201c;Norm. Prec.&#x201d; is an abbreviation for normalized precision.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Tracker</th>
<th align="center">Success</th>
<th align="center">Precision</th>
<th align="center">Norm. Prec.</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">ATOM [<xref ref-type="bibr" rid="B30">30</xref>]</td>
<td align="center">0.703</td>
<td align="center">0.648</td>
<td align="center">0.771</td>
</tr>
<tr>
<td align="center">SiamRPN&#x2b;&#x2b;</td>
<td align="center">0.733</td>
<td align="center">0.649</td>
<td align="center">0.800</td>
</tr>
<tr>
<td align="center">SiamAttn</td>
<td align="center">0.752</td>
<td align="center">-</td>
<td align="center">0.81</td>
</tr>
<tr>
<td align="center">SiamFC&#x2b;&#x2b; [<xref ref-type="bibr" rid="B31">31</xref>]</td>
<td align="center">0.754</td>
<td align="center">0.705</td>
<td align="center">0.800</td>
</tr>
<tr>
<td align="center">AutoMatch [<xref ref-type="bibr" rid="B32">32</xref>]</td>
<td align="center">0.760</td>
<td align="center">0.726</td>
<td align="center">-</td>
</tr>
<tr>
<td align="center">TrSiam [<xref ref-type="bibr" rid="B33">33</xref>]</td>
<td align="center">0.781</td>
<td align="center">0.721</td>
<td align="center">0.829</td>
</tr>
<tr>
<td align="center">TrDiMP</td>
<td align="center">0.784</td>
<td align="center">0.731</td>
<td align="center">0.833</td>
</tr>
<tr>
<td align="center">
<bold>Ours</bold>
</td>
<td align="center">
<bold>0.788</bold>
</td>
<td align="center">
<bold>0.754</bold>
</td>
<td align="center">
<bold>0.836</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The meaning of the bold values in table primarily represents the training results of the methods proposed in this article.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s4-3-4">
<title>4.3.4 Evaluation on the OTB100 dataset</title>
<p>OTB100 is a classical benchmark for visual object tracking and contains 100 short-term videos, with an average of 590 frames per video. The results of our proposed algorithm against other algorithms on the OTB100 dataset are shown in <xref ref-type="fig" rid="F7">Figure 7</xref>, and the results demonstrate that our method outperforms other algorithms in terms of success rate metrics (AUC) as well as accuracy. In addition, to be able to visualize the actual tracking effect of the object tracking algorithm, we selected four representative video frames from the 100 videos of the OTB100 dataset for visualization, which contain most of the challenges encountered in object tracking scenarios. As shown in <xref ref-type="table" rid="T4">Table 4</xref>, these four sequences selected include the following challenges: scale variation, deformation, object partial occlusion, and background interference<italic>.</italic> By using different colored object tracking frames to represent the tracking effect of the algorithm in this paper and other compared mainstream object tracking algorithms (SiamRPN&#x2b;&#x2b;, SiamCAR, and PACNet) in the same image (<xref ref-type="fig" rid="F8">Figure 8</xref>), it is possible to analyze more intuitively these algorithms in the same image (<xref ref-type="fig" rid="F8">Figure 8</xref>), allowing a more visual analysis of the tracking performance of these algorithms. From the visualization results, the algorithm proposed in this paper shows long-term stability in tracking sequences under complex challenges. The results of the first line and the fourth line of the video sequences are mainly related to the challenges of cluttered backgrounds as well as scale variations, which shows that our tracker maintains a high level of robustness, while the other three methods drift during the tracking process. Tests from the second row of video sequences show that our method exhibits stable and consistent tracking performance when the object undergoes rotational changes. The information obtained from the third row of video sequences shows that our model can accurately track the object even in the presence of object occlusions and recognize the object accurately when it reappears. By visualizing the tracking process as described previously, we again validate the high tracking performance of our model.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Test results of our method on the OTB100 dataset. <bold>(A)</bold> Precise plot and <bold>(B)</bold> success plot.</p>
</caption>
<graphic xlink:href="fphy-11-1269638-g007.tif"/>
</fig>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Video frame challenge attributes for the selected OTB100 dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Sequence name</th>
<th align="left">Frame number</th>
<th align="left">Challenge attribute</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Soccer</td>
<td align="left">35,165,215</td>
<td align="left">IV, SV, OCC, MB, FM, IPR, OPR, BC</td>
</tr>
<tr>
<td align="left">Toy</td>
<td align="left">171,216,271</td>
<td align="left">IV, IPR, OPR</td>
</tr>
<tr>
<td align="left">Jogging</td>
<td align="left">18,56,84</td>
<td align="left">OCC, DEF, OPR</td>
</tr>
<tr>
<td align="left">Skating1</td>
<td align="left">103,182,253</td>
<td align="left">OCC, DEF, OPR</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Comparison of our proposed tracker with some advanced trackers on four challenging OTB100 video sequences; the results show that our approach can effectively address these challenges. Reproduced from the OTB100 dataset, [<xref ref-type="bibr" rid="B20">20</xref>], with permission from IEEE.</p>
</caption>
<graphic xlink:href="fphy-11-1269638-g008.tif"/>
</fig>
</sec>
<sec id="s4-3-5">
<title>4.3.5 Evaluation on the WATB dataset</title>
<p>The WATB dataset is a common wildlife dataset containing over 203,000 frames and 206 video sequences covering a wide range of animals on land, in the ocean, and in the sky. The average length of the videos was over 980 frames. Each video was manually labeled with 13 challenge attributes, including partial occlusion, rotation, and deformation. All frames in the dataset were labeled with axis-aligned bounding boxes. To test the performance of our model on this dataset, we tested our method using WATB, and the tracking accuracy and success plots are shown in <xref ref-type="fig" rid="F9">Figure 9</xref> and <xref ref-type="fig" rid="F10">Figure 10</xref>, respectively. According to the results of the tracking success rate curve and precision rate curve, our method outperforms other algorithms and achieves optimal performance when compared to other tracking methods. In addition, we also tested the performance against some challenging attributes, including rotation, deformation, partial occlusion, and scale variation. According to the results shown in <xref ref-type="fig" rid="F11">Figure 11</xref>, our method shows continuous better performance among the compared trackers.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Precision plot of our method on the WATB dataset.</p>
</caption>
<graphic xlink:href="fphy-11-1269638-g009.tif"/>
</fig>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>Success plot of our method on the WATB dataset.</p>
</caption>
<graphic xlink:href="fphy-11-1269638-g010.tif"/>
</fig>
<fig id="F11" position="float">
<label>FIGURE 11</label>
<caption>
<p>Some of the challenges tested on the WATB dataset include out-of-plane rotation, in-plane rotation, deformation, and partial occlusion. <bold>(A, B, E, F)</bold> Results of precision plots and <bold>(C, D, G, H)</bold> results of success plots.</p>
</caption>
<graphic xlink:href="fphy-11-1269638-g011.tif"/>
</fig>
</sec>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>In this paper, a novel tracking framework based on the spatio-temporal interactive fusion network is proposed. Considering the spatio-temporal correlation between frames, a feature enhancement network is used to process both memory and query branches by combining historical frame information, and a spatio-temporal interactive fusion network is proposed to achieve effective filtering and fusion of feature information of the two branches, which improves the generalization ability of the network and makes full use of contextual information. In the feature enhancement network, by introducing a spatio-temporal feature enhancement network, the memory frame features are enhanced in the temporal dimension as well as the spatial dimension, and the query frame features are enhanced only in the spatial dimension, enabling the tracker to locate the object more accurately. The method proposed in this paper can cope with most complex situations, but the problem of object loss for small objects and for situations, where there are more similar objects interfering in the background, still exists, while the method can be improved in other ways. Overall, through extensive experimental results on the GOT-10k, OTB100, TrackingNet, LaSOT, and WATB datasets, the tracking method proposed in this paper shows better performance.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/Supplementary Material; further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s7">
<title>Author contributions</title>
<p>DH: conceptualization and writing&#x2013;original draft. SY: investigation, software, visualization, writing&#x2013;original draft, and writing&#x2013;review and editing. JD: resources and writing&#x2013;review and editing. YW: investigation, methodology, and writing&#x2013;original draft. AY: project administration, resources, and writing&#x2013;original draft. YW: software and writing&#x2013;review and editing. JX: visualization and writing&#x2013;original draft.</p>
</sec>
<sec id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This research was funded by the National Natural Science Foundation of China (grant number 62127813) and the Science and Technology Key Research and Development Project of Jilin Province (grant number 20230201071GX).</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>F</given-names>
</name>
<name>
<surname>He</surname>
<given-names>J</given-names>
</name>
</person-group>. <article-title>Learning saliency-aware correlation filters for visual tracking</article-title>. <source>Comp J</source> (<year>2022</year>) <volume>65</volume>(<issue>7</issue>):<fpage>1846</fpage>&#x2013;<lpage>59</lpage>. <pub-id pub-id-type="doi">10.1093/comjnl/bxab026</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Bertinetto</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Valmadre</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Henriques</surname>
<given-names>JF</given-names>
</name>
<name>
<surname>Vedaldi</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Torr</surname>
<given-names>PHS</given-names>
</name>
</person-group>. <article-title>Fully convolutional siamese networks for object tracking</article-title>. <conf-name>Proceedings of the computer vision &#x2013; ECCV 2016 workshops</conf-name>, <conf-loc>Amsterdam, Netherlands</conf-loc>, <conf-date>October 2016</conf-date>. </citation>
</ref>
<ref id="B3">
<label>3.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Gonzalez-Garcia</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Weijer</surname>
<given-names>J</given-names>
</name>
</person-group>, <article-title>Learning the model update for siamese trackers</article-title>, <conf-name>Proceedings of the IEEE/CVF international conference on computer vision</conf-name>. <conf-loc>Seoul, Korea (South)</conf-loc>, <conf-date>October 2019</conf-date>: <fpage>4010</fpage>&#x2013;<lpage>9</lpage>.</citation>
</ref>
<ref id="B4">
<label>4.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>W</given-names>
</name>
</person-group>, <article-title>Deformable siamese attention networks for visual object tracking</article-title>, <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>. <conf-loc>Seattle, WA, USA</conf-loc>, <conf-date>June 2020</conf-date>: <fpage>6728</fpage>&#x2013;<lpage>37</lpage>.</citation>
</ref>
<ref id="B5">
<label>5.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>B</given-names>
</name>
</person-group>, <article-title>Siamese tracking network with multi-attention mechanism[J]</article-title> (<year>2023</year>). <pub-id pub-id-type="doi">10.21203/rs.3.rs-3296460/v1</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>F</given-names>
</name>
</person-group>. <article-title>Deblurring transformer tracking with conditional cross-attention</article-title>. <source>Multimedia Syst</source> (<year>2023</year>) <volume>29</volume>(<issue>3</issue>):<fpage>1131</fpage>&#x2013;<lpage>44</lpage>. <pub-id pub-id-type="doi">10.1007/s00530-022-01043-0</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>Q</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>W</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>C</given-names>
</name>
</person-group> <article-title>Learning dynamic siamese network for visual object tracking</article-title>, <conf-name>Proceedings of the IEEE international conference on computer vision</conf-name>. <conf-loc>Venice, Italy</conf-loc>, <conf-date>October 2017</conf-date>: <fpage>1763</fpage>&#x2013;<lpage>71</lpage>.</citation>
</ref>
<ref id="B8">
<label>8.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Yan</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>H</given-names>
</name>
<etal/>
</person-group> <article-title>Learning spatio-temporal transformer for visual tracking</article-title>, <conf-name>Proceedings of the IEEE/CVF international conference on computer vision</conf-name>. <conf-loc>Montreal, BC, Canada</conf-loc>, <conf-date>October 2021</conf-date>: <fpage>10448</fpage>&#x2013;<lpage>57</lpage>.</citation>
</ref>
<ref id="B9">
<label>9.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Fu</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Q</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>Z</given-names>
</name>
</person-group> <article-title>Stmtrack: template-free visual tracking with space-time memory networks</article-title>, <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>. <conf-loc>Nashville, TN, USA</conf-loc>, <conf-date>June 2021</conf-date>: <fpage>13774</fpage>&#x2013;<lpage>83</lpage>.</citation>
</ref>
<ref id="B10">
<label>10.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Kuai</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Porikli</surname>
<given-names>F</given-names>
</name>
</person-group>. <article-title>End-to-end feature integration for correlation filter tracking with channel attention</article-title>. <source>IEEE Signal Process. Lett</source> (<year>2018</year>) <volume>25</volume>(<issue>12</issue>):<fpage>1815</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1109/lsp.2018.2877008</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rahman</surname>
<given-names>MM</given-names>
</name>
<name>
<surname>Fiaz</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Jung</surname>
<given-names>SK</given-names>
</name>
</person-group>. <article-title>Efficient visual tracking with stacked channel-spatial attention learning</article-title>. <source>IEEE Access</source> (<year>2020</year>) <volume>8</volume>:<fpage>100857</fpage>&#x2013;<lpage>69</lpage>. <pub-id pub-id-type="doi">10.1109/access.2020.2997917</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>X</given-names>
</name>
</person-group>, <article-title>A twofold siamese network for real-time object tracking</article-title>, <conf-name>Proceedings of the IEEE conference on computer vision and pattern recognition</conf-name>. <conf-loc>Salt Lake City, UT, USA</conf-loc>, <conf-date>June 2018</conf-date>: <fpage>4834</fpage>&#x2013;<lpage>43</lpage>.</citation>
</ref>
<ref id="B13">
<label>13.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Tian</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H</given-names>
</name>
</person-group>, <article-title>Fcos: fully convolutional one-stage object detection</article-title>, <conf-name>Proceedings of the IEEE/CVF international conference on computer vision</conf-name>. <conf-loc>Seoul, Korea (South)</conf-loc>, <conf-date>October 2019</conf-date>: <fpage>9627</fpage>&#x2013;<lpage>36</lpage>.</citation>
</ref>
<ref id="B14">
<label>14.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>TY</given-names>
</name>
<name>
<surname>Goyal</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Girshick</surname>
<given-names>R</given-names>
</name>
</person-group>, <article-title>Focal loss for dense object detection</article-title>, <conf-name>Proceedings of the IEEE international conference on computer vision</conf-name>. <conf-loc>Venice, Italy</conf-loc>, <conf-date>October 2017</conf-date>: <fpage>2980</fpage>&#x2013;<lpage>8</lpage>.</citation>
</ref>
<ref id="B15">
<label>15.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Muller</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Bibi</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Giancola</surname>
<given-names>S</given-names>
</name>
</person-group>, <article-title>Trackingnet: a large-scale dataset and benchmark for object tracking in the wild</article-title>, <conf-name>Proceedings of the European conference on computer vision (ECCV)</conf-name>. <conf-loc>Munich, Germany</conf-loc>, <conf-date>September 2018</conf-date>: <fpage>300</fpage>&#x2013;<lpage>17</lpage>.</citation>
</ref>
<ref id="B16">
<label>16.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Fan</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>F</given-names>
</name>
</person-group>, <article-title>Lasot: a high-quality benchmark for large-scale single object tracking</article-title>, <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>. <conf-loc>Long Beach, CA, USA</conf-loc>, <conf-date>June 2019</conf-date>: <fpage>5374</fpage>&#x2013;<lpage>83</lpage>.</citation>
</ref>
<ref id="B17">
<label>17.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>K</given-names>
</name>
</person-group>. <article-title>GOT-10k: a large high-diversity benchmark for generic object tracking in the wild</article-title>. <source>IEEE Trans pattern Anal machine intelligence</source> (<year>2019</year>) <volume>43</volume>(<issue>5</issue>):<fpage>1562</fpage>&#x2013;<lpage>77</lpage>. <pub-id pub-id-type="doi">10.1109/tpami.2019.2957464</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Russakovsky</surname>
<given-names>O</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Krause</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Satheesh</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>S</given-names>
</name>
<etal/>
</person-group> <article-title>Imagenet large scale visual recognition challenge</article-title>. <source>Int J Comput Vis</source> (<year>2015</year>) <volume>115</volume>:<fpage>211</fpage>&#x2013;<lpage>52</lpage>. <pub-id pub-id-type="doi">10.1007/s11263-015-0816-y</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>TY</given-names>
</name>
<name>
<surname>Maire</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Belongie</surname>
<given-names>S</given-names>
</name>
</person-group>, <article-title>Microsoft coco: common objects in context</article-title>, <conf-name>Proceedings of the computer vision&#x2013;ECCV 2014: 13th European conference</conf-name>, <conf-loc>Zurich, Switzerland</conf-loc>, <conf-date>September, 2014</conf-date>: <fpage>740</fpage>&#x2013;<lpage>55</lpage>.</citation>
</ref>
<ref id="B20">
<label>20.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Lim</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>MH</given-names>
</name>
</person-group>. <article-title>Object tracking benchmark</article-title>. <source>IEEE Trans pattern Anal machine intelligence</source> (<year>2015</year>) <volume>37</volume>(<issue>9</issue>):<fpage>1834</fpage>&#x2013;<lpage>48</lpage>. <pub-id pub-id-type="doi">10.1109/tpami.2014.2388226</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X</given-names>
</name>
<name>
<surname>He</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>F</given-names>
</name>
</person-group>. <article-title>WATB: wild animal tracking benchmark</article-title>. <source>Int J Comp Vis</source> (<year>2023</year>) <volume>131</volume>(<issue>4</issue>):<fpage>899</fpage>&#x2013;<lpage>917</lpage>. <pub-id pub-id-type="doi">10.1007/s11263-022-01732-3</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22.</label>
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Paszke</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Gross</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Chintala</surname>
<given-names>S</given-names>
</name>
</person-group>, <article-title>Automatic differentiation in pytorch</article-title> (<year>2017</year>). <ext-link ext-link-type="uri" xlink:href="https://openreview.net/pdf?id=BJJsrmfCZ#:%7E:text=PyTorch%2C%20like%20most%20other%20deep,which%20usually%20differentiate%20a%20single">https://openreview.net/pdf?id&#x3d;BJJsrmfCZ&#x23;:&#x223c;:text&#x3d;PyTorch%2C%20like%20most%20other%20deep,which%20usually%20differentiate%20a%20single</ext-link>.</citation>
</ref>
<ref id="B23">
<label>23.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>do Carmo Fran&#xe7;a</surname>
<given-names>HF</given-names>
</name>
<name>
<surname>Soares</surname>
<given-names>A</given-names>
</name>
</person-group>. <source>GoogLeNet-going deeper with convolutions</source> (<year>2014</year>). <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1409.4842">https://arxiv.org/abs/1409.4842</ext-link>.</citation>
</ref>
<ref id="B24">
<label>24.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>24Zhang</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>Z</given-names>
</name>
</person-group>. <article-title>UAST: uncertainty-aware siamese tracking</article-title>, <conf-name>Proceedings of the international conference on machine learning. PMLR</conf-name>, <conf-loc>Baltimore, Maryland, USA</conf-loc>, <conf-date>July 2022</conf-date>.</citation>
</ref>
<ref id="B25">
<label>25.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Shao</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>Y</given-names>
</name>
</person-group>, <article-title>Graph attention tracking</article-title>, <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>. <conf-loc>Nashville, TN, USA</conf-loc>, <conf-date>June 2021</conf-date>: <fpage>9543</fpage>&#x2013;<lpage>52</lpage>.</citation>
</ref>
<ref id="B26">
<label>26.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>W</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Q</given-names>
</name>
</person-group>, <article-title>Siamrpn&#x2b;&#x2b;: evolution of siamese visual tracking with very deep networks</article-title>, <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>. <conf-loc>Long Beach, CA, USA</conf-loc>, <conf-date>June 2019</conf-date>.</citation>
</ref>
<ref id="B27">
<label>27.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Bhat</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Danelljan</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Gool</surname>
<given-names>LV</given-names>
</name>
</person-group>, <article-title>Learning discriminative model prediction for tracking</article-title>, <conf-name>Proceedings of the IEEE/CVF international conference on computer vision</conf-name>. <conf-loc>Seoul, Korea (South)</conf-loc>, <conf-date>October 2019</conf-date>: <fpage>6182</fpage>&#x2013;<lpage>91</lpage>.</citation>
</ref>
<ref id="B28">
<label>28.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>J</given-names>
</name>
</person-group>, <article-title>Ocean: object-aware anchor-free tracking</article-title>, <conf-name>Proceedings of the computer vision&#x2013;ECCV 2020: 16th European conference</conf-name>, <conf-loc>Glasgow, UK</conf-loc>, <conf-date>August, 2020</conf-date>.</citation>
</ref>
<ref id="B29">
<label>29.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>R</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>M</given-names>
</name>
</person-group>. <article-title>Visual tracking via hierarchical deep reinforcement learning</article-title>, <conf-name>Proceedings of the AAAI Conf Artif Intelligence</conf-name>. <conf-loc>Washington DC, USA</conf-loc>, <conf-date>February, 2021</conf-date>, <pub-id pub-id-type="doi">10.1609/aaai.v35i4.16443</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Danelljan</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Bhat</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Khan</surname>
<given-names>FS</given-names>
</name>
</person-group>, <article-title>Atom: accurate tracking by overlap maximization</article-title>, <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>. <conf-loc>Long Beach, CA, USA</conf-loc>, <conf-date>June 2019</conf-date>: <fpage>4660</fpage>&#x2013;<lpage>9</lpage>.</citation>
</ref>
<ref id="B31">
<label>31.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>F</given-names>
</name>
</person-group>, <article-title>Siamfc&#x2b;&#x2b;: towards robust and accurate visual tracking with object estimation guidelines</article-title>, <conf-name>Proceedings of the AAAI conference on artificial intelligence</conf-name>. <conf-loc>New York, NY, USA</conf-loc>, <conf-date>February 2020</conf-date>: <fpage>12549</fpage>&#x2013;<lpage>56</lpage>.</citation>
</ref>
<ref id="B32">
<label>32.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X</given-names>
</name>
</person-group>, <article-title>Learn to match: automatic matching network design for visual tracking</article-title>, <conf-name>Proceedings of the IEEE/CVF international conference on computer vision</conf-name>, <conf-loc>Montreal, BC, Canada</conf-loc>, <conf-date>October 2021</conf-date>: <fpage>13339</fpage>&#x2013;<lpage>48</lpage>.</citation>
</ref>
<ref id="B33">
<label>33.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>W</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J</given-names>
</name>
</person-group>, <article-title>Transformer meets tracker: exploiting temporal context for robust visual tracking</article-title>, <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>. <conf-loc>Nashville, TN, USA</conf-loc>, <conf-date>June 2021</conf-date>: <fpage>1571</fpage>&#x2013;<lpage>80</lpage>.</citation>
</ref>
</ref-list>
</back>
</article>