<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Robot. AI</journal-id>
<journal-title>Frontiers in Robotics and AI</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Robot. AI</abbrev-journal-title>
<issn pub-type="epub">2296-9144</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1490718</article-id>
<article-id pub-id-type="doi">10.3389/frobt.2024.1490718</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Robotics and AI</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Embedding-based pair generation for contrastive representation learning in audio-visual surveillance data</article-title>
<alt-title alt-title-type="left-running-head">Wang et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/frobt.2024.1490718">10.3389/frobt.2024.1490718</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Wang</surname>
<given-names>Wei-Cheng</given-names>
</name>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2786867/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>De Coninck</surname>
<given-names>Sander</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/2927535/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Leroux</surname>
<given-names>Sam</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/2596624/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Simoens</surname>
<given-names>Pieter</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/564537/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff>
<institution>IDLab</institution>, <institution>Ghent University&#x2014;imec</institution>, <addr-line>Ghent</addr-line>, <country>Belgium</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1034986/overview">Anoop Cherian</ext-link>, Mitsubishi Electric Research Laboratories (MERL), United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1504175/overview">Saadane Rachid</ext-link>, &#xc9;cole Hassania des Travaux Publics, Morocco</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1822719/overview">Aili Wang</ext-link>, Harbin University of Science and Technology, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2583135/overview">Sarah Samson Juan</ext-link>, University of Malaysia Sarawak, Malaysia</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Wei-Cheng Wang, <email>weicheng.wang@ugent.be</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>13</day>
<month>01</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>11</volume>
<elocation-id>1490718</elocation-id>
<history>
<date date-type="received">
<day>03</day>
<month>09</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>09</day>
<month>12</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Wang, De Coninck, Leroux and Simoens.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Wang, De Coninck, Leroux and Simoens</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Smart cities deploy various sensors such as microphones and RGB cameras to collect data to improve the safety and comfort of the citizens. As data annotation is expensive, self-supervised methods such as contrastive learning are used to learn audio-visual representations for downstream tasks. Focusing on surveillance data, we investigate two common limitations of audio-visual contrastive learning: false negatives and the minimal sufficient information bottleneck. Irregular, yet frequently recurring events can lead to a considerable number of false-negative pairs and disrupt the model&#x2019;s training. To tackle this challenge, we propose a novel method for generating contrastive pairs based on the distance between embeddings of different modalities, rather than relying solely on temporal cues. The semantically synchronized pairs can then be used to ease the minimal sufficient information bottleneck along with the new loss function for multiple positives. We experimentally validate our approach on real-world data and show how the learnt representations can be used for different downstream tasks, including audio-visual event localization, anomaly detection, and event search. Our approach reaches similar performance as state-of-the-art modality- and task-specific approaches.</p>
</abstract>
<kwd-group>
<kwd>self-supervised learning</kwd>
<kwd>surveillance</kwd>
<kwd>audio-visual representation learning</kwd>
<kwd>contrastive learning</kwd>
<kwd>audio-visual event localization</kwd>
<kwd>anomaly detection</kwd>
<kwd>event search</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Robot Vision and Artificial Perception</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Today, around 55 percent of the global population is living in an urban area or city, and this number is expected to rise to 68 percent by 2050 (<xref ref-type="bibr" rid="B14">United Nations Department of Economic and Social Affairs, 2018</xref>). To support this urbanization in a sustainable way, smart cities deploy a variety of sensor, networking and data analysis technologies to improve their operational efficiency and safety measures. Cameras and microphones are two prevalent sensors in smart city applications. Cameras primarily serve surveillance functions, facilitating crime prevention and traffic monitoring, while microphones are utilized for detecting phenomena such as gunshots or glass shattering (<xref ref-type="bibr" rid="B36">Mydlarz et al., 2017</xref>). Deploying cameras and microphones in the same location enables more comprehensive situational insights. Audio and video cues provide complementary information, which enhances the robustness of event detection against challenges encountered in real-world settings, including noise, occlusions, or low-light conditions (<xref ref-type="bibr" rid="B7">Bajovic et al., 2021</xref>).</p>
<p>Deep neural networks are currently the state-of-the-art solution for audio-visual surveillance tasks such as vehicle detection (<xref ref-type="bibr" rid="B32">Mao et al., 2020</xref>), violent scene detection (<xref ref-type="bibr" rid="B49">Ullah et al., 2023</xref>), and sound tagging (<xref ref-type="bibr" rid="B6">Bai et al., 2022</xref>). However, training these models requires large (labelled) datasets that are expensive to collect. Furthermore, research indicates the advantages of employing location-specific models for surveillance (<xref ref-type="bibr" rid="B28">Leroux et al., 2022</xref>), further increasing the amount of training data and associated labels that need to be collected.</p>
<p>The objective of this work is to design a scalable framework for learning representations of real-world audio-visual surveillance data in a self-supervised manner. The resulting representations should generalize well to a wide range of downstream surveillance tasks, meaning that the training of task-specific models that will take these representations as input requires little or no labelled data. Examples of downstream tasks for smart city surveillance include event localization (<xref ref-type="bibr" rid="B39">Ran et al., 2022</xref>), anomaly detection (<xref ref-type="bibr" rid="B52">Wu et al., 2022</xref>; <xref ref-type="bibr" rid="B26">Kumari and Saini, 2022</xref>; <xref ref-type="bibr" rid="B27">Leporowski et al., 2023</xref>) and event search (<xref ref-type="bibr" rid="B35">Munjal et al., 2019</xref>).</p>
<p>Self-supervised learning of transferable representations is typically achieved by training a feature extraction model on a pretext task. Contrastive learning, a specific type of self-supervised learning, formulates the training objective in terms of a distance metric between the representations of a pair of input samples. The goal is to minimize the distance for semantically similar instances (positive pairs) and maximize it for dissimilar instances (negative pairs). The process of generating positive and negative pairs during training is a crucial factor in obtaining transferable features. Negative pairs are often generated through random sampling from the dataset. Positive pairs can be constructed without requiring label information by pairing a sample with an augmented version of that sample. Such augmentations are straightforward in the case of static images, but much harder to design for temporal data (<xref ref-type="bibr" rid="B38">Qian et al., 2021</xref>). In the case of multi-modal data, positive pairs can be naturally formed by treating audio and video clips sampled at the same timestamp within a stream as positive pairs, a pair generation mechanisms known as Audio-Visual Synchronization (AVS) (<xref ref-type="bibr" rid="B5">Aytar et al., 2016</xref>). To distinguish it from our approach, we refer to it as Temporal-based Pair Generation (TPG) to highlight that typical Audio-Visual Synchronization takes temporal cues when generating data pairs.</p>
<p>TPG however introduces two challenges related to the semantic repetition that is observed in audio-visual surveillance data over time. First, a large temporal distance between two fragments of a recording does not guarantee a semantical difference between these fragments. Ambulances, police cars, buses, auditory beacons for visually impaired pedestrians, or vans with similar appearance are only a handful examples of scenes recurring at unpredictable and variable intervals. One example taken from a surveillance camera in Tokyo is shown in <xref ref-type="fig" rid="F1">Figures 1A, B</xref> shows two visually and aurally similar trucks appearing at different time frames. When sampling a data pair where the visual modality is taken from (A) and the audio modality from (B), this pair is labeled as negative based on the time stamps, despite their semantic similarity. Such mislabeled pairs, referred to as <italic>false negatives</italic>, compromise the training process and cause the learned embedding spaces to lose the semantic meaning (<xref ref-type="bibr" rid="B60">Zolfaghari et al., 2021</xref>; <xref ref-type="bibr" rid="B43">Sun et al., 2023</xref>; <xref ref-type="bibr" rid="B12">Chuang et al., 2020</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Sampling data from different timestamps may result in false negative pairs if both timestamps share a similar audio-visual context. <bold>(A)</bold> Visual frame at 00:31:52. <bold>(B)</bold> Visual frame at 01:39:40.</p>
</caption>
<graphic xlink:href="frobt-11-1490718-g001.tif"/>
</fig>
<p>Another limitation of relying on temporal cues to generate positive and negative pairs in contrastive learning arises from the information bottleneck in the training objective. Since all supervision information for learning a representation of one element comes from the contrasting element (<xref ref-type="bibr" rid="B45">Tian et al., 2020a</xref>), the representations are <italic>minimal sufficient</italic>, meaning that they are focused on the mutual information between the samples of positive pairs. While this is effective when the downstream task is aligned with the pretext task, the minimal sufficient may not contain enough information to generalize across multiple downstream tasks (<xref ref-type="bibr" rid="B47">Tsai et al., 2021</xref>; <xref ref-type="bibr" rid="B50">Wang et al., 2022</xref>; <xref ref-type="bibr" rid="B16">Feichtenhofer et al., 2021</xref>). Increasing the number of positive pairs for each sample can address this limitation, as it makes the pretext task more challenging and encourages the representations to encode richer information (<xref ref-type="bibr" rid="B12">Chuang et al., 2020</xref>; <xref ref-type="bibr" rid="B25">Khosla et al., 2020</xref>; <xref ref-type="bibr" rid="B45">Tian et al., 2020a</xref>). With TPG, each video clip forms a single positive pair with its corresponding audio, leading to minimal sufficient representations that only contain information on objects that are simultaneously audible and visible. However, real-world events are complex, involving multiple elements at different time intervals, as shown in <xref ref-type="fig" rid="F2">Figure 2</xref>. <xref ref-type="fig" rid="F2">Figures 2A&#x2013;C</xref> shows the progressive stages of a police car cautiously passing by a busy intersection. Due to the relative distance and velocity between the camera and the vehicles, as well as interactions between vehicles, each timestamp is a unique combination of visual and audio cues of the same event. Because of the information bottleneck, the representation of this scene might not contain all the visual information (police car and vehicle stopping) with its audio modality (siren). In the case of <xref ref-type="fig" rid="F2">Figure 2B</xref>, when learning a minimal sufficient representation, part of the visual information could be ignored (e.g., vehicle stopping), although this information could be essential for downstream tasks. For example, during an emergency, the audio of the siren and the visual cue of other vehicles stopping are crucial for managing the traffic light before the police car enters the camera&#x2019;s view. Meanwhile, <xref ref-type="fig" rid="F2">Figure 2D</xref> shows a similar event occuring at another time, where a police car approaches from a different road. By creating positive pairs using samples from both events, the learned representations will contain more comprehensive information, accommodating the complexities of real-world scenarios.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Example stills illustrating two instances of a police car passing an intersection with activated siren, forcing other vehicles to stop. The police car is indicated in yellow. Stopped and moving cars are indicated with red signs and green arrows respectively. <bold>(A)</bold> Only the siren of the police car is audible. <bold>(B)</bold> The police car enters the scene. <bold>(C)</bold> The police car leaves the scene but siren is still audible. <bold>(D)</bold> On a later time, a police car enters the scene with a different view angle.</p>
</caption>
<graphic xlink:href="frobt-11-1490718-g002.tif"/>
</fig>
<p>To reduce false negatives as well as to learn representations with sufficient information, semantically similar events should be mapped together in the embedding space. In this paper, we introduce the <italic>Embedding-based Pair Generation</italic> (<italic>EPG</italic>) mechanism as an alternative for TPG to sample positive or negative pairs. Our approach detects false negative pairs by calculating a distance between the embeddings of two instances of the <italic>same</italic> modality. Furthermore, we propose a new loss that considers multiple positives simultaneously to learn representations that contain more information, further improving the transferability of the learned features to a variety of downstream tasks. We train a pseudo-Siamese network to encode the audio segments and video frames to the same embedding space. After training, the model has learnt what audio typically corresponds to certain visual inputs and <italic>vice versa</italic>. The two deep neural networks can then serve as feature extractors, jointly or separately, for downstream tasks.</p>
<p>To summarize, our main contributions are as follows:<list list-type="simple">
<list-item>
<p>1. We identify the inherent flaws in applying the widely used audio-visual correspondence to smart surveillance data. An embedding-based pair generation is introduced to tackle this problem;</p>
</list-item>
<list-item>
<p>2. We study the limitation of minimal sufficient representation for audio-visual representation learning in surveillance. We then propose a novel loss to encode richer task-relevant information to improve the performance on downstream tasks;</p>
</list-item>
<list-item>
<p>3. We evaluate our approach with supervised downstream tasks and demonstrate the effectiveness of our improvement comparing to the-state-of-the-art approaches on audio-visual representation learning. We further qualitatively evaluate our approach on two unsupervised tasks applied on real-world surveillance data.</p>
</list-item>
</list>
</p>
<p>The remainder of this paper is structured as follows. <xref ref-type="sec" rid="s2">Section 2</xref> provides the related work of audio-visual representation learning, self-supervised learning, and how positives and negatives are generated for contrastive learning. In <xref ref-type="sec" rid="s3">Section 3</xref>, we describe the mechanism of embedding-based pair selection and how we incorporate multiple positives in the contrastive loss. The downstream tasks and datasets used to evaluate the learnt representation are explained in <xref ref-type="sec" rid="s4">Section 4</xref>. We then describe the implementation details and the discussion on the experimental results in <xref ref-type="sec" rid="s5">Section 5</xref>. We conclude in <xref ref-type="sec" rid="s6">Section 6</xref> and give some directions for further research.</p>
</sec>
<sec id="s2">
<title>2 Related work</title>
<p>Our work lies at the intersection of three domains: audio-visual representation learning, self-supervised representation learning, and pair generation for contrastive learning. In the following subsections, we provide an overview of the approaches in each of these fields that are most pertinent to our work.</p>
<sec id="s2-1">
<title>2.1 Audio-visual representation learning</title>
<p>The analysis of audio-visual data is gaining popularity as audio and visual information offer complementary insights on the same content. The two modalities are expected to align, either at the frame level or the instance level. Jointly considering both modalities benefits the analysis of audio-visual tasks such as active speaker detection (<xref ref-type="bibr" rid="B3">Afouras et al., 2020b</xref>), sound source localization (<xref ref-type="bibr" rid="B58">Zhou et al., 2023</xref>), lip reading (<xref ref-type="bibr" rid="B2">Afouras et al., 2020a</xref>), or video forensics (<xref ref-type="bibr" rid="B17">Feng et al., 2023</xref>).</p>
<p>To encode audio-visual representation, most frameworks employ three components: an audio encoder, a visual encoder, and a projector. This modular design accounts for the distinct characteristics of audio and visual data, requiring different processing configurations, such as varying network architectures and learning schedules. Once high-level information is extracted by the encoders, the embeddings are connected by passing them through the projector. This setup is especially useful for tasks like anomaly event recognition (<xref ref-type="bibr" rid="B18">Gao et al., 2024</xref>), where labelled data is easier to obtain, allowing for straightforward end-to-end training to learn audio-visual representations.</p>
<p>However, in most real-world scenarios, audio-visual data is collected continuously, making it challenging to obtain annotations. To address this, self-supervised learning (SSL) is a promising approach to leverage the semantic synchronization between audio and video. The inherent correlation between audio and visual elements serves as a natural indicator when designing the pretext task in self-supervised learning. When applying SSL to audio-visual data, the general concept involves training a model to differentiate between matching and non-matching pairs of video segments and audio excerpts. Negative pairs are constructed either by sampling audio and video from different recordings, known as audio-visual correspondence (AVC), or by sampling from different offsets in the same recording, termed audio-visual synchronization (AVS).</p>
<p>AVC as pretext task was first introduced by <xref ref-type="bibr" rid="B4">Arandjelovic and Zisserman (2017)</xref>, who demonstrated that the learnt representations obtained competitive results on both audio tasks, such as sound classification, and visual tasks, such as image classification and object detection. Subsequent works extended AVC to tasks such as action recognition (<xref ref-type="bibr" rid="B34">Morgado et al., 2021</xref>), active speaker detection (<xref ref-type="bibr" rid="B3">Afouras et al., 2020b</xref>) or sound source localization (<xref ref-type="bibr" rid="B58">Zhou et al., 2023</xref>). More recently, <xref ref-type="bibr" rid="B23">Huang et al. (2024)</xref> proposed a hybrid approach combining generative SSL objectives with contrastive learning. With a joint loss function, both inter-modal and intra-modal relationships can be considered by the model.</p>
<p>AVS, as a more nuanced pretext task, leverages the temporal synchronization between audio and video to pretrain the model. This approach has been successfully applied to tasks such as lip reading (<xref ref-type="bibr" rid="B2">Afouras et al., 2020a</xref>), video forensics (<xref ref-type="bibr" rid="B17">Feng et al., 2023</xref>) and active speaker detection (<xref ref-type="bibr" rid="B54">Wuerkaixi et al., 2022</xref>).</p>
</sec>
<sec id="s2-2">
<title>2.2 Self-supervised representation learning</title>
<p>While supervised learning has made a great achievement in many research domains, accessing reliable annotations for data is sometimes expensive or impractical. Self-supervised learning aims to learn a representative embedding by leveraging the information within the data instead of relying on the supervision of annotations. SSL introduces pretext tasks, which are auxiliary tasks designed to train the model to learn representations that can later be applied to downstream tasks. These pretext tasks may not directly relate to the target task but serve as an effective means of extracting generalizable embeddings.</p>
<p>According to the type of the pretext task, SSL approaches can be divided into three different categories (<xref ref-type="bibr" rid="B51">Wang Y. et al., 2022</xref>): generative, predictive and contrastive. Below, we briefly describe all three and will then focus on the contrastive approaches as these form the basis for our work.</p>
<p>Generative SSL employs generative models, such as AutoEncoders (AEs) or Generative Adversarial Networks (GANs), coupled with pixel-level reconstruction loss functions to learn representative features. This approach is particularly popular in the field of computer vision (<xref ref-type="bibr" rid="B21">He et al., 2022</xref>; <xref ref-type="bibr" rid="B52">Wu et al., 2022</xref>; <xref ref-type="bibr" rid="B53">Wu et al., 2023</xref>). While pixel-level reconstruction is an intuitive and effective pretext task, the generative SSL methods can be hard to train. During the training of the generative model, model tend to focus overly on background details at the expense of the foreground content. This happens particularly when the foreground content is relatively small in terms of frame ratio, known as the foreground-background imbalance. Another common challenge is the object scale imbalance, where the size of the objects varies when the camera has a more oblique view, as discussed in (<xref ref-type="bibr" rid="B41">Sampath et al., 2021</xref>). Both problems require additional mechanisms to focus on specific semantic information.</p>
<p>Predictive SSL methods utilize self-generated labels derived from predefined transformations of the input data to guide network training. Pretext tasks such as classifying rotated versions of the original image (<xref ref-type="bibr" rid="B19">Gidaris et al., 2018</xref>), or arranging image regions within a jigsaw puzzle (<xref ref-type="bibr" rid="B33">Misra and Maaten, 2020</xref>) have been demonstrated to result in high-level features of images. These pretext tasks preserve the semantic meaning of the content. It is however not trivial to design good pretext tasks for temporal data, e.g., in surveillance applications, due to the added complexity of sequence dynamics.</p>
<p>Finally, contrastive SSL aims to overcome the challenges encountered in the aforementioned approaches. Contrastive SSL compares pairs of data samples to learn representations that maximize similarity for positive pairs (semantically similar samples) and minimize similarity for negative pairs (semantically dissimilar samples). Positive pairs typically consist of different augmentations of the same input sample (<xref ref-type="bibr" rid="B11">Chen et al., 2020</xref>; <xref ref-type="bibr" rid="B55">Zbontar et al., 2021</xref>), while negative pairs are created by randomly pairing samples from the dataset. Both inputs are projected to a shared embedding space. By training the model to maximize the mutual information between embeddings of samples in positive pairs and minimize that of samples in negative pairs, the model learns to extract high-level features that can be used for downstream tasks. Contrastive SSL extends to multimodal data by forming pairs with samples from each modality. For instance, in audiovisual representation learning, video fragments paired with corresponding audio fragments represent positive pairs, while combinations of video frames and randomly selected audio snippets serve as negative pairs (<xref ref-type="bibr" rid="B43">Sun et al., 2023</xref>; <xref ref-type="bibr" rid="B39">Ran et al., 2022</xref>). Text is another modality commonly used in conjunction with video, in particular to learn language-video representations by pairing the video with its caption (<xref ref-type="bibr" rid="B60">Zolfaghari et al., 2021</xref>; <xref ref-type="bibr" rid="B57">Zhang et al., 2023</xref>).</p>
<p>While contrastive learning has been proven to be effective in many applications, there are two notable limitations, namely, <italic>false negatives</italic> (<xref ref-type="bibr" rid="B43">Sun et al., 2023</xref>) and the minimal sufficient representation (<xref ref-type="bibr" rid="B45">Tian et al., 2020a</xref>) problem. False negatives occur when semantically similar pairs are mistakenly labelled as negative due to the design of the pretext task. <xref ref-type="bibr" rid="B60">Zolfaghari et al. (2021)</xref> describes the impact of false negatives and proposes identifying influential samples. These samples, which are more likely to be false negative samples, have high feature similarity with other samples and should be removed. Similarly, <xref ref-type="bibr" rid="B43">Sun et al. (2023)</xref> proposes a statistical approach to locate false negatives by considering the similarity between the same modality of different sample pairs. The information bottleneck leading to minimal sufficient representation, containing information that is sufficient for the pretext task, is rather rare in scholarly discussions. <xref ref-type="bibr" rid="B45">Tian et al. (2020a)</xref> thoroughly describes the concept with theoretical and empirical proof, stating that when the downstream tasks are not aligned with the pretext task, it might downgrade the representative of learnt features. This issue is especially pronounced when applying contrastive learning to surveillance data, where events unfold over time and involve multiple stages. A minimal sufficient representation may not be able to describe the different stages of an event. To improve the usability of learned representations in downstream tasks on such data, several works aim to incorporate multiple positives in the objective function (<xref ref-type="bibr" rid="B12">Chuang et al., 2020</xref>; <xref ref-type="bibr" rid="B25">Khosla et al., 2020</xref>; <xref ref-type="bibr" rid="B45">Tian et al., 2020a</xref>). By considering multiple semantically related positives in the objective function, models can better capture the diverse aspects, improving the richness and generality of the learnt features.</p>
</sec>
<sec id="s2-3">
<title>2.3 Pair generation for contrastive learning</title>
<p>The selection of positive and negative pairs is a crucial factor in contrastive learning (<xref ref-type="bibr" rid="B22">He et al., 2020</xref>; <xref ref-type="bibr" rid="B42">Shah et al., 2022</xref>). Many works assume that randomly selected inputs lack semantic similarity and can be used as negative pairs. However, this assumption can introduce false negative pairs. This issue, a phenomenon also referred to as <italic>sampling bias</italic>, has been shown to hinder performance. <xref ref-type="bibr" rid="B12">Chuang et al. (2020)</xref> empirically demonstrate significant performance gains across multiple research domains when false negatives are avoided. Other recent studies prove theoretically and empirically that the quality of the negative samples is more important than their quantity (<xref ref-type="bibr" rid="B24">Kalantidis et al., 2020</xref>; <xref ref-type="bibr" rid="B59">Zhu et al., 2021</xref>).</p>
<p>Recent works have explored several strategies to refine the process of generating positive and negative data pairs. At the level of instance sampling, <xref ref-type="bibr" rid="B24">Kalantidis et al. (2020)</xref> enhance training efficiency and the quality of the learned representations by synthesizing hard negatives. These hard negatives closely resemble positive pairs and are therefore challenging for the model to distinguish. During the training process, novel hard negatives are synthesized as feature-level linear combinations of the currently hardest examples. <xref ref-type="bibr" rid="B46">Tian et al. (2020b)</xref> focus on enhancing the diversity of the sampled positive pairs. They argue that improving the diversity in positive pairs helps the model learn representations that are invariant to nuisance variables, since the representations are focused on the mutual information across all positive views. <xref ref-type="bibr" rid="B59">Zhu et al. (2021)</xref> introduce a feature transformation technique that manipulates features to create both hard positives and diverse negatives. Beyond improving the pair selection, <xref ref-type="bibr" rid="B12">Chuang et al. (2020)</xref> propose the <italic>debiased contrastive loss</italic>, a novel training objective function that considers the approximated distribution of negative samples instead of relying on explicit negative samples.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Proposed method</title>
<p>In the following sections, we will first explain the different components of the framework. Then, we introduce the novel embedding-based pair generation (<italic>EPG</italic>) mechanism designed to reduce the number of false negatives. Finally, we introduce a novel loss function and elaborate on how this loss function might address the challenge of minimal sufficient representations.</p>
<sec id="s3-1">
<title>3.1 Architecture</title>
<p>As illustrated in <xref ref-type="fig" rid="F3">Figure 3</xref>, the audio-visual representation learning block follows a pseudo-Siamese structure with two encoders: <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Both encoders are deep convolutional networks, designed to process visual and audio information, respectively. Where in conventional Siamese architectures, the parameters are shared between the encoders, here the encoders have a different structure, hence the name <italic>pseudo</italic>-Siamese.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Overview of the framework. A pair of video clip <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and audio segment <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> are served as the input of a two stream pseudo-Siamese network. The network consists of a visual encoder <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and an audio encoder <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to project the input into the same embedding space. A embedding-based label of the data pair is calculated to determine whether the data pair is positive or negative. These labels are later used to compute the proposed loss and to update the network. After the pseudo-Siamese network is trained, both encoders can be used as a feature extractor jointly or separately for downstream tasks.</p>
</caption>
<graphic xlink:href="frobt-11-1490718-g003.tif"/>
</fig>
<p>The encoders project the audio and video onto a shared embedding space <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:mi mathvariant="script">Z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Surveillance data <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is first split into short clips, where each clip <inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> comprises the sequence of frames <inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and an audio segment <inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. By selecting from the frames and segments of <inline-formula id="inf12">
<mml:math id="m12">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, we first generate a data pair <inline-formula id="inf13">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, consisting of the <inline-formula id="inf14">
<mml:math id="m14">
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th video segment and the <inline-formula id="inf15">
<mml:math id="m15">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th audio fragment. With <inline-formula id="inf16">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf17">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf18">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is encoded into <inline-formula id="inf19">
<mml:math id="m19">
<mml:mrow>
<mml:mi mathvariant="script">Z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, yielding <inline-formula id="inf20">
<mml:math id="m20">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. By utilizing <inline-formula id="inf21">
<mml:math id="m21">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and the pair generation mechanism, a contrastive loss <inline-formula id="inf22">
<mml:math id="m22">
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is calculated to train the network. Once <inline-formula id="inf23">
<mml:math id="m23">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf24">
<mml:math id="m24">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are trained, the encoders can be used as feature extractors for downstream tasks, either jointly or independently.</p>
</sec>
<sec id="s3-2">
<title>3.2 Embedding-based pair generation</title>
<p>Recognizing the limitations of relying solely on time offsets to ascertain semantic dissimilarity, we introduce an alternative solution to identify temporally non-aligned but semantically similar data pairs with the distance in the embedding space.</p>
<p>The set of possible pairs is denoted as <inline-formula id="inf25">
<mml:math id="m25">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Instead of solely relying on the condition <inline-formula id="inf26">
<mml:math id="m26">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to label the pair <inline-formula id="inf27">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> pair as positive, we propose to calculate the mutual information between <inline-formula id="inf28">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf29">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in the embedding space <inline-formula id="inf30">
<mml:math id="m30">
<mml:mrow>
<mml:mi mathvariant="script">Z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. The mutual information <inline-formula id="inf31">
<mml:math id="m31">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> can be computed using a distance metric <inline-formula id="inf32">
<mml:math id="m32">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="script">Z</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> such as Euclidean distance or cosine similarity (<xref ref-type="bibr" rid="B10">Boudiaf et al., 2020</xref>). If the mutual information between the video or audio embeddings, <inline-formula id="inf33">
<mml:math id="m33">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> or <inline-formula id="inf34">
<mml:math id="m34">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, is higher than a threshold <inline-formula id="inf35">
<mml:math id="m35">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, we say <inline-formula id="inf36">
<mml:math id="m36">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf37">
<mml:math id="m37">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are semantically similar, even though they are recorded at different times.</p>
<p>To identify whether the two elements of the pair <inline-formula id="inf38">
<mml:math id="m38">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are semantically similar, we calculate the label <inline-formula id="inf39">
<mml:math id="m39">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> with the following equation:<disp-formula id="e1">
<mml:math id="m40">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="matrix">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="script">Z</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mo>&#x2228;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="script">Z</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="script">Z</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3e;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mo>&#x2227;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="script">Z</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3e;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<p>Intuitively, this definition ascertains that when two audio fragments are semantically similar <inline-formula id="inf40">
<mml:math id="m41">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2248;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, we assume that the corresponding video fragments in time also should be semantically similar, and <italic>vice versa</italic>. Conversely, two video fragments close in embedding space are hypothesized to have semantically close audio fragments. Hence, positive pairs can be constructed by mixing modalities of <inline-formula id="inf41">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf42">
<mml:math id="m43">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. With mutual information between the video or audio embeddings, we can further identify the false negative pairs and consider them as positive. Consequently, these positive pairs can also be used to reduce the limitation of minimal sufficient representation.</p>
</sec>
<sec id="s3-3">
<title>3.3 Contrastive loss with multi-positive pairs</title>
<p>Given that the encoders of both modalities are trained to predict similar feature representations for temporally aligned audio and video clips, they tend to focus on information present in both modalities. Contrastive loss guides the training of the encoders such that the video and audio embedding are close: <inline-formula id="inf43">
<mml:math id="m44">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2248;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. This means that <inline-formula id="inf44">
<mml:math id="m45">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> learns to eliminate all information not present in <inline-formula id="inf45">
<mml:math id="m46">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, and <italic>vice versa</italic>. As explained in the introduction, other audio segments may contain complementary information, but this will be eliminated in the representation of <inline-formula id="inf46">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>The main purpose of including multiple modalities however, is to complement each other, providing additional information when an object or person can not be observed in one of the modalities. This is particularly a problem for data pair generation that only relies on temporal alignment since it allows only one positive pair for each time frame:<disp-formula id="e2">
<mml:math id="m48">
<mml:mrow>
<mml:mo>&#x2200;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo>,</mml:mo>
<mml:mo>&#x2203;</mml:mo>
<mml:mo>!</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>To address this limit of minimal sufficient representation, we propose a modification to the conventional contrastive loss function that uses multiple positive pairs identified through embedding-based distance <xref ref-type="disp-formula" rid="e1">Equation 1</xref>.</p>
<p>Different from temporal-based pair generation, the embedding-based pair generation mechanism does not solely rely on temporal information to create positive pairs with <inline-formula id="inf47">
<mml:math id="m49">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>. With the semantic similarity of the embedding space, the limitation due to <xref ref-type="disp-formula" rid="e2">Equation 2</xref> may be reduced as there can be multiple positives for <inline-formula id="inf48">
<mml:math id="m50">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>:<disp-formula id="e3">
<mml:math id="m51">
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mfenced open="{" close="}">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
<p>The upper bound on the mutual information between <inline-formula id="inf49">
<mml:math id="m52">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and the union of all elements in the set in 3 is higher than the mutual information between <inline-formula id="inf50">
<mml:math id="m53">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf51">
<mml:math id="m54">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>. As a result, including multiple positives will likely retain more information on <inline-formula id="inf52">
<mml:math id="m55">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> in the representation <inline-formula id="inf53">
<mml:math id="m56">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, which will benefit downstream task performance. Inspired by the loss function proposed in (<xref ref-type="bibr" rid="B20">Hadsell et al., 2006</xref>), we proposed a modified loss function to fit the multi-positive found by <xref ref-type="disp-formula" rid="e3">Equation 3</xref>.<disp-formula id="e4">
<mml:math id="m57">
<mml:mrow>
<mml:mtable class="aligned">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">EPG</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mfenced open="(" close="">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="script">Z</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">x</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="script">Z</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
</p>
<p>The distance metric <inline-formula id="inf54">
<mml:math id="m58">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="script">Z</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> measures the distance between <inline-formula id="inf55">
<mml:math id="m59">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf56">
<mml:math id="m60">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> in the shared space <inline-formula id="inf57">
<mml:math id="m61">
<mml:mrow>
<mml:mi mathvariant="script">Z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, which is bounded by a predefined constant <inline-formula id="inf58">
<mml:math id="m62">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> in the second term of <xref ref-type="disp-formula" rid="e4">Equation 4</xref>. Note that, due to the symmetry between each modality of two semantically similar data pairs, <xref ref-type="disp-formula" rid="e4">Equation 4</xref> is equivalent to pairing each visual modality <inline-formula id="inf59">
<mml:math id="m63">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> within the dataset with each audio <inline-formula id="inf60">
<mml:math id="m64">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. The distance function <inline-formula id="inf61">
<mml:math id="m65">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="script">Z</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> can represent any similarity metric. In this paper, we obtained the best results using a weighted combination of the Euclidean distance <inline-formula id="inf62">
<mml:math id="m66">
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and the cosine similarity <inline-formula id="inf63">
<mml:math id="m67">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. As cosine similarity yields larger values for higher similarity; the distance function <inline-formula id="inf64">
<mml:math id="m68">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="script">Z</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is defined as follows:<disp-formula id="e5">
<mml:math id="m69">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="script">Z</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x22c5;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>where <inline-formula id="inf65">
<mml:math id="m70">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a value between 0 and 1 to control the ratio between the two distance functions. The use of <inline-formula id="inf66">
<mml:math id="m71">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is discussed in <xref ref-type="sec" rid="s5-4">Section 5.4</xref>.</p>
</sec>
</sec>
<sec id="s4">
<title>4 Experimental setup</title>
<p>In this section, we describe the experiments conducted to evaluate the quality of the learned representations across various downstream tasks relevant to smart surveillance applications. The evaluation includes one supervised downstream task, namely, audio-visual event detection, and two unsupervised tasks: anomaly detection and event query. For all tasks, the pseudo-Siamese network is first pretrained on the audio-visual data using the objective function of <xref ref-type="disp-formula" rid="e4">Equation 4</xref> in a self-supervised manner. Subsequently, the weights of the audio and/or visual encoders were frozen and considered as fixed feature extractors while training a small network for each of the downstream tasks.</p>
<sec id="s4-1">
<title>4.1 Implementation details</title>
<p>For all experiments, we maintain consistent configurations for data preprocessing, network architecture, and training procedure. Task-specific variations are described in subsequent sections. The audiovisual dataset is first segmented into 1-s, non-overlapping recordings, each containing synchronized audio segments and video frames.</p>
<p>Video recordings are downsampled to 5 fps, and each frame is resized to <inline-formula id="inf67">
<mml:math id="m72">
<mml:mrow>
<mml:mn>398</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>224</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> pixels. To align with the visual encoder&#x2019;s input specifications, each frame is further divided into two overlapping <inline-formula id="inf68">
<mml:math id="m73">
<mml:mrow>
<mml:mn>224</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>224</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> crops. These two crops are treated as independent frames that map to the same audio segment, and are then jointly considered during the inference phase.</p>
<p>Audio segments are resampled to 44,100 Hz and transformed into log-mel spectrograms, following the configuration outlined by <xref ref-type="bibr" rid="B1">Adapa (2019)</xref>. This transformation utilizes a window size of 256, a hop length of 694, and a total of 128 bins.</p>
<p>The framework consists of a pseudo-Siamese network with a visual encoder <inline-formula id="inf69">
<mml:math id="m74">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and an audio encoder <inline-formula id="inf70">
<mml:math id="m75">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf71">
<mml:math id="m76">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is built based on X3D-M (<xref ref-type="bibr" rid="B15">Feichtenhofer, 2020</xref>), featuring one convolutional layer, four residual blocks, and a final classification layer. We take the implementation from PyTorchVideo but replace its last layer with a fully-connected layer with 512 neurons to align with the dimensions of the audio representations. <inline-formula id="inf72">
<mml:math id="m77">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is implemented as a ResNet18 model, taking the log-mel coefficients as input. The parameter counts for <inline-formula id="inf73">
<mml:math id="m78">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf74">
<mml:math id="m79">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are 4.02 million and 4.16 million, respectively. Training specifics differ: <inline-formula id="inf75">
<mml:math id="m80">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> trains with a learning rate of <inline-formula id="inf76">
<mml:math id="m81">
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>e</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> and a decay of <inline-formula id="inf77">
<mml:math id="m82">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>e</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, while <inline-formula id="inf78">
<mml:math id="m83">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is trained with a learning rate of <inline-formula id="inf79">
<mml:math id="m84">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>e</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> and a decay of <inline-formula id="inf80">
<mml:math id="m85">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>e</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>For <inline-formula id="inf81">
<mml:math id="m86">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, we take the pre-trained weights of X3D-M, which is trained on Kinetics-400, provided by PyTorchVideo. As for <inline-formula id="inf82">
<mml:math id="m87">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, we train it from scratch by freezing the pre-trained <inline-formula id="inf83">
<mml:math id="m88">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and exclusively training <inline-formula id="inf84">
<mml:math id="m89">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Subsequently, using a layer-wise learning approach (<xref ref-type="bibr" rid="B8">Belilovsky et al., 2019</xref>), both encoders undergo iterative training.</p>
<p>The embedding space <inline-formula id="inf85">
<mml:math id="m90">
<mml:mrow>
<mml:mi mathvariant="script">Z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is designed to capture semantically meaningful information, which may not be guaranteed when training from scratch. To ensure robust initialization, we only consider a pair as positive when the two modalities are sampled from the same timestamps during the first epoch of training. In all subsequent epochs, training shifts to the embedding-based label <inline-formula id="inf86">
<mml:math id="m91">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, guided by the loss function <inline-formula id="inf87">
<mml:math id="m92">
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Throughout training, the weight parameter <inline-formula id="inf88">
<mml:math id="m93">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> for the distance function in <xref ref-type="disp-formula" rid="e5">Equation 5</xref> is kept constant at 2.5, and <inline-formula id="inf89">
<mml:math id="m94">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is set to 0. An ablation study of these parameters is provided in 5.4. As <inline-formula id="inf90">
<mml:math id="m95">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the threshold to determine whether the distance in embedding space of a temporally misaligned data pair is smaller than a temporally aligned pair, we set the <inline-formula id="inf91">
<mml:math id="m96">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> as the distance between the embeddings of the temporally aligned pair <inline-formula id="inf92">
<mml:math id="m97">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="script">Z</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. Thus, the threshold <inline-formula id="inf93">
<mml:math id="m98">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> adapts dynamically during training based on the embedding space.</p>
</sec>
<sec id="s4-2">
<title>4.2 Supervised tasks</title>
<p>After pretraining the audio and video encoders, they can be used as feature extractors for downstream tasks. The first task we consider is event localization, which aims to pinpoint specific predefined events within an audiovisual stream, such as the entry of a car. This task can be cast as a supervised learning problem by following the protocol outlined in (<xref ref-type="bibr" rid="B39">Ran et al., 2022</xref>). Long recordings are divided into short clips, and the objective is to perform binary classification to determine whether a given clip contains the target event.</p>
<sec id="s4-2-1">
<title>4.2.1 Dataset</title>
<p>Finding real-world surveillance datasets that contain video, audio and labels is challenging. Some established datasets, such as those used in <xref ref-type="bibr" rid="B9">Benfold and Reid (2011)</xref>; <xref ref-type="bibr" rid="B40">Ristani et al. (2016)</xref>, have been taken down due to privacy concerns. Other datasets primarily consist of very short clips gathered from diverse locations, often sourced from video-sharing platforms like YouTube (<xref ref-type="bibr" rid="B5">Aytar et al., 2016</xref>; <xref ref-type="bibr" rid="B37">Perez et al., 2019</xref>; <xref ref-type="bibr" rid="B13">Danesh Pazho et al., 2023</xref>). While these datasets suffice for certain supervised tasks, such as violence detection, they fall short for our purpose of learning features in a semi-supervised manner over long audiovisual streams.</p>
<p>We decided to use the Toulouse Campus Surveillance Dataset (ToCaDa) (<xref ref-type="bibr" rid="B31">Malon et al., 2020</xref>) to validate our approach. The ToCaDa dataset encompasses two distinct scenarios, each captured by multiple cameras strategically positioned to record simultaneously audio and video. Some cameras have overlapping fields of view. The events in the videos are scripted to demonstrate a possible burglary involving 20 actors playing roles as pedestrians or suspects. Each video has an approximate duration of 300 s and comes with detailed annotations for both audio and video events. Most videos in this dataset contain only a limited number of events, making a meaningful split of each video across train and test set difficult. However, scenario one of the ToCaDa dataset contains a higher number of cameras observing the same scene, including some with slightly different viewpoints, see <xref ref-type="fig" rid="F4">Figure 4</xref>. We therefore use the footage of Camera 2 as the training set for learning representations, and evaluate event localization on the audiovisual recordings from all other cameras.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Camera setup for ToCaDa Scenario 1. The camera view used for <inline-formula id="inf94">
<mml:math id="m99">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is marked in red, while the camera views that are used for <inline-formula id="inf95">
<mml:math id="m100">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are labelled as yellow. The rest of the camera views are used as <inline-formula id="inf96">
<mml:math id="m101">
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Note that both similar sets and challenging sets are used for testing.</p>
</caption>
<graphic xlink:href="frobt-11-1490718-g004.tif"/>
</fig>
</sec>
<sec id="s4-2-2">
<title>4.2.2 Evaluation procedure</title>
<p>We evaluate the transferability of the learned representations to this supervised classification task by adopting the linear evaluation protocol from <xref ref-type="bibr" rid="B50">Wang et al. (2022)</xref>. After pretraining, the weights of the feature extractors are frozen and a one-layer linear classifier is trained using cross-entropy loss as the objective function. After training the classifier, we evaluate performance using a segment-wise classification accuracy matrix, again following the protocol presented in (<xref ref-type="bibr" rid="B39">Ran et al., 2022</xref>).</p>
</sec>
<sec id="s4-2-3">
<title>4.2.3 Baseline methods</title>
<p>We first evaluate our method by comparing it with existing audio-visual representation learning methods, as well as the <italic>TPG</italic> baseline. Additionally, since the data contains both audio and video, we compare the classification results of our method with visual-only and audio-only methods.</p>
<p>For multi-modal audio-visual representation learning, we compare our method with TACMA and MAVil. TACMA (<xref ref-type="bibr" rid="B39">Ran et al., 2022</xref>) is a self-supervised representation learning technique specifically designed for audio-visual event localization. TACMA employs a Barlow-Twins architecture to learn representations and includes a cross-modal attention module to enhance audio-visual information capture. However, because the cross-modal attention module is trained in a supervised manner, we exclude it and use only the AV-BT module to generate the representations. MAViL (<xref ref-type="bibr" rid="B23">Huang et al., 2024</xref>) is a recent method that has demonstrated strong performance in event classification tasks across general audio-visual datasets.</p>
<p>Both TACMA and MAViL are designed for datasets containing short videos with diverse content and scenarios. To ensure compatability with their training scheme, we segment the long ToCaDa training videos into 10-s, non-overlapping subclips. Unlike TACMA, MAViL relies on negative pairs to compute inter-modal contrastive loss. To enable this, we pair the audio and video from different subclips to create negative pairs for MAViL.</p>
<p>For the <italic>TPG</italic> baseline, we use the same training procedure as our method, with the only difference being the pair generation strategy.</p>
<p>For visual-only benchmarks, we employ the EfficientNet (<xref ref-type="bibr" rid="B44">Tan and Le, 2019</xref>) and X3D (<xref ref-type="bibr" rid="B15">Feichtenhofer, 2020</xref>) models. EfficientNet is pretrained on the ImageNet dataset, and we select the EfficientNet-B0 variant, which has 4.03 million parameters, making it comparable in scale to the <inline-formula id="inf97">
<mml:math id="m102">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> encoder in our method. For X3D, our choice is the X3D-M variant pretrained on Kinetics-400, which contains 3.76 million parameters. Both models and their pretrained weights are sourced from the PyTorch and PyTorchVideo repositories. To use these models as baselines, we remove their final classification layers and use the outputs of the remaining pretrained network as representations.</p>
<p>TACMA and X3D-M process input frames at a resolution of <inline-formula id="inf98">
<mml:math id="m103">
<mml:mrow>
<mml:mn>256</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>256</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, while EfficientNet-b0 operates at <inline-formula id="inf99">
<mml:math id="m104">
<mml:mrow>
<mml:mn>224</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>224</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. For fair comparison, we first downsize all frames to <inline-formula id="inf100">
<mml:math id="m105">
<mml:mrow>
<mml:mn>224</mml:mn>
<mml:mspace width="0.3333em"/>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mn>224</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> and then upsample them to <inline-formula id="inf101">
<mml:math id="m106">
<mml:mrow>
<mml:mn>256</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>256</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> for use with TACMA and X3D-M.</p>
<p>For all other configurations, we follow the original preprocessing steps specified in the respective works, except for MAViL. Since the authors of MAViL did not release their code or the pretrained model, we follow the implementation details and the pretrained model of the reproduction reported in (<xref ref-type="bibr" rid="B48">Tseng et al., 2024</xref>).</p>
<p>For the audio-only benchmarks, we use the best-performing model from the DCASE19 urban sound tagging task (<xref ref-type="bibr" rid="B1">Adapa, 2019</xref>). The official implementation and pretrained weights were obtained from the author&#x2019;s GitHub repository<xref ref-type="fn" rid="fn1">
<sup>1</sup>
</xref>. Similarly to the visual-only benchmarks, we remove the classification layer from the pretrained network and use its output as feature representation for downstream evaluation.</p>
</sec>
</sec>
<sec id="s4-3">
<title>4.3 Unsupervised tasks</title>
<p>We also evaluate our framework in two unsupervised tasks commonly used in surveillance: anomaly detection and query-guided event search. The task of anomaly detection involves identifying inputs that deviate from normal behavior. Since the behaviors of interest are not defined beforehand, anomaly detection is a challenging task that requires high-quality input features to discern subtle deviations. The query-guided event search task is to locate events in a video similar to a given query event. For instance, if the query is a clip containing a joyriding car with distinct audio or visual characteristics, the task is to identify other timestamps in the recording where similar events occur.</p>
<sec id="s4-3-1">
<title>4.3.1 Dataset</title>
<p>The ToCaDa dataset, while valuable for tasks like event localization, is less suited for anomaly detection and query-guided event search due to its limited video length and restricted diversity of actions occuring. Similarly, widely-used datasets with annotated anomalies, such as Avenue (<xref ref-type="bibr" rid="B30">Lu et al., 2013</xref>) and ShangHaiTech (<xref ref-type="bibr" rid="B29">Liu et al., 2018</xref>), contain only visual cues. As an alternative, we collected real-world audio-visual surveillance footage from a publicly available live stream on YouTube<xref ref-type="fn" rid="fn2">
<sup>2</sup>
</xref>,<xref ref-type="fn" rid="fn3">
<sup>3</sup>
</xref>. This audiovisual stream captures a main intersection in Tokyo&#x2019;s Shinjuku district, observed from a a high vantage point, providing a representative setting for urban surveillance. Some stills from the recordings are shown in <xref ref-type="fig" rid="F1">Figure 1</xref>, showcasing different types of vehicles, bikes and pedestrians with their accompanying sounds. For in-depth evaluation, we recorded four 4-hour-long videos from two different dates under different lighting conditions. The videos are recorded during two time windows: 15:00 to 19:00 (daytime) and 19:00 to 23:00 (nighttime), on a Tuesday and a Thursday.</p>
</sec>
<sec id="s4-3-2">
<title>4.3.2 Evaluation procedure</title>
<p>Since the Tokyo dataset is not annotated, we conduct a qualitative evaluation between different methods of anomaly detection and query-guided event search.</p>
<p>The anomaly score for a clip <inline-formula id="inf102">
<mml:math id="m107">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is calculated using the distance function of <xref ref-type="disp-formula" rid="e5">Equation 5</xref>. A clip is flagged as anomalous if the score is higher than a threshold. For each 4-hour-long video, a separate model is trained to learn audio-visual representations. The threshold for anomaly detection is dynamically adapted for each video and set to <inline-formula id="inf103">
<mml:math id="m108">
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> where <inline-formula id="inf104">
<mml:math id="m109">
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf105">
<mml:math id="m110">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represent the mean and standard deviation of the anomaly score on the training set. This threshold considers <inline-formula id="inf106">
<mml:math id="m111">
<mml:mrow>
<mml:mn>95.45</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of the training data as normal.</p>
<p>For query-guided event search, separate models are trained on each video to extract features. Query events have been selected manually. Events were searched in all recordings, but events within a window of 1 min before and after the selected event are excluded as search results. We rank the search results based on the distance between the query and the results in the embedding space.</p>
</sec>
<sec id="s4-3-3">
<title>4.3.3 Baseline methods</title>
<p>For the anomaly detection task, we compare our method with five other approaches: two multi-modal fusion models (<xref ref-type="bibr" rid="B26">Kumari and Saini, 2022</xref>) (<italic>Fusion</italic>) and (<xref ref-type="bibr" rid="B23">Huang et al., 2024</xref>) (MAViL), one vision-only approach based on the X3D-M model, one audio-only baseline Adapa (<xref ref-type="bibr" rid="B1">Adapa, 2019</xref>) and the multi-modal <italic>TPG</italic> baseline.</p>
<p>As for the event search task, apart from the multi-modal <italic>TPG</italic> and MAViL (<xref ref-type="bibr" rid="B23">Huang et al., 2024</xref>) baselines, we compare our approach to a baseline that involves a straightforward fusion approach in which video and audio features of separately trained encoders are concatenated. Specifically, we concatenate visual features from X3D-M (<xref ref-type="bibr" rid="B15">Feichtenhofer, 2020</xref>) with audio features from Adapa (<xref ref-type="bibr" rid="B1">Adapa, 2019</xref>). We refer to this baseline as (<italic>A &#x2b; V</italic>).</p>
</sec>
</sec>
</sec>
<sec id="s5">
<title>5 Experimental results</title>
<sec id="s5-1">
<title>5.1 Audio-visual event localization</title>
<p>The experimental results for audio-visual event localization are summarized in <xref ref-type="table" rid="T1">Table 1</xref>, which reports the accuracy score for each method. The ToCaDa dataset consists of videos recorded from 18 different cameras. We trained on data from camera 2 and tested on all other cameras. To better assess the robustness of the learned features, we categorize the test cameras into two groups: (1) cameras with a view similar to the training camera (labeled as &#x201c;similar&#x201d;), and (2) cameras with distinct perspectives compared to the training camera (labeled as &#x201c;challenging&#x201d;). By reporting results separately, we ensure that the evaluation reflects the model&#x2019;s generalization ability without artificially inflating accuracy due to overlapping cameras.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Audio-visual event localization results on two subsets of ToCaDa dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="center">Method</th>
<th align="center">&#x23; Params.(M)</th>
<th align="center">Similar</th>
<th align="center">Challenging</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="4" align="left">Audio-Visual</td>
<td align="left">TACMA (<xref ref-type="bibr" rid="B39">Ran et al., 2022</xref>)</td>
<td align="right">150.46</td>
<td align="right">77.41</td>
<td align="right">67.20</td>
</tr>
<tr>
<td align="left">MAViL (<xref ref-type="bibr" rid="B23">Huang et al., 2024</xref>)</td>
<td align="right">185.66</td>
<td align="right">72.69</td>
<td align="right">62.34</td>
</tr>
<tr>
<td align="left">
<italic>TPG</italic>
</td>
<td align="right">15.2</td>
<td align="right">71.33</td>
<td align="right">60.47</td>
</tr>
<tr>
<td align="left">
<italic>EPG</italic> (Ours)</td>
<td align="right">15.2</td>
<td align="right">86.92</td>
<td align="right">77.48</td>
</tr>
<tr>
<td rowspan="6" align="left">Visual</td>
<td align="left">EfficientNet<xref ref-type="table-fn" rid="Tfn1">
<sup>a</sup>
</xref>
</td>
<td align="right">4.01</td>
<td align="right">85.65</td>
<td align="right">73.20</td>
</tr>
<tr>
<td align="left">X3D-M<xref ref-type="table-fn" rid="Tfn1">
<sup>a</sup>
</xref>
</td>
<td align="right">2.97</td>
<td align="right">90.21</td>
<td align="right">71.84</td>
</tr>
<tr>
<td align="left">TACMA (<xref ref-type="bibr" rid="B39">Ran et al., 2022</xref>)</td>
<td align="right">52.87</td>
<td align="right">62.15</td>
<td align="right">43.60</td>
</tr>
<tr>
<td align="left">MAViL (<xref ref-type="bibr" rid="B23">Huang et al., 2024</xref>)</td>
<td align="right">85.74</td>
<td align="right">65.23</td>
<td align="right">51.44</td>
</tr>
<tr>
<td align="left">
<italic>TPG</italic>
</td>
<td align="right">4.02</td>
<td align="right">60.71</td>
<td align="right">44.92</td>
</tr>
<tr>
<td align="left">
<italic>EPG</italic> (Ours)</td>
<td align="right">4.02</td>
<td align="right">85.42</td>
<td align="right">70.02</td>
</tr>
<tr>
<td rowspan="5" align="left">Audio</td>
<td align="left">Adapa<xref ref-type="table-fn" rid="Tfn1">
<sup>a</sup>
</xref> (<xref ref-type="bibr" rid="B1">Adapa, 2019</xref>)</td>
<td align="right">4.16</td>
<td align="right">85.22</td>
<td align="right">82.47</td>
</tr>
<tr>
<td align="left">TACMA (<xref ref-type="bibr" rid="B39">Ran et al., 2022</xref>)</td>
<td align="right">97.58</td>
<td align="right">76.43</td>
<td align="right">65.61</td>
</tr>
<tr>
<td align="left">MAViL (<xref ref-type="bibr" rid="B23">Huang et al., 2024</xref>)</td>
<td align="right">85.74</td>
<td align="right">77.24</td>
<td align="right">62.30</td>
</tr>
<tr>
<td align="left">
<italic>TPG</italic>
</td>
<td align="right">4.16</td>
<td align="right">70.66</td>
<td align="right">60.43</td>
</tr>
<tr>
<td align="left">
<italic>EPG</italic> (Ours)</td>
<td align="right">4.16</td>
<td align="right">81.37</td>
<td align="right">79.48</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="Tfn1">
<label>
<sup>a</sup>
</label>
<p>Denotes the model is pretrained on a large-scale dataset without any fine-tuning. The numbers are segment-wise classification accuracy following the protocol in TACMA (<xref ref-type="bibr" rid="B39">Ran et al., 2022</xref>).</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The results highlight a significant performance gap between the two camera sets, with all methods achieving considerably higher accuracy on the &#x201c;similar&#x201d; set. These results corroborate findings in earlier research on the advantages of using location-specific methods for analyzing surveillance data (<xref ref-type="bibr" rid="B28">Leroux et al., 2022</xref>).</p>
<p>When comparing our approach with the other audio-visual techniques TACMA and MAViL, our approach demonstrates a substantial performance advantage. This difference can likely be attributed to the fact that these models are designed for more general purposes that require training on large-scale datasets with diverse content. Due to the more constrained nature of the ToCaDa dataset, these models struggle to learn sufficiently representative embeddings for event localization tasks. Examining the performance difference between TACMA and MAViL, we observe that while MAViL achieves great performance in general representation learning, its reliance on both reconstruction and contrastive loss with negative pairs defined based on temporal alignment, pose limitations in this dataset. In contrast, TACMA employs a Barlow-Twins architecture, which avoids the need for negative pairs, and only consider limited positive pairs.</p>
<p>Notably, when our model architecture is trained using the <italic>TPG</italic> pair generation strategy instead of the <italic>EPG</italic> strategy, a significant drop in accuracy is observed. This results highlights the advantages of accounting for semantic similarity in pair selection.</p>
<p>We also compare the classification results obtained using the video representations learned by our approach, TACMA and MAViL, with those from video-only based models. Our approach performs similar to the EfficientNet model, which was pretrained on the large scale ImageNet dataset. The best performance is achieved by the pretrained X3D-M model, which is expected given its architecture&#x2019;s specific design for capturing motion information. In contrast, the self-supervised methods TACMA, MAViL, and <italic>TPG</italic> perform poorly in this task. MAViL&#x2019;s weaker performance can be attributed to its reliance on reconstruction loss, which is susceptible to foreground-background imbalance. This imbalance makes MAViL more sensitive to differences in the scene. These results demonstrate that the feature representations obtained from our multimodal approach also transfer effectively to purely visual tasks.</p>
<p>For audio-only event localization, we observe a smaller drop in accuracy between the &#x201c;similar&#x201d; and &#x201c;challenging&#x201d; locations compared to video-only localization. This underscores the robustness of audio data, which is less sensitive to the exact positioning of sensors. The best results in this setting are obtained by the pretrained Adapa model. While our model performs slightly worse than Adapa, it achieves this without the need for pretraining on extensive labeled data. Furthermore, our proposed <italic>EPG</italic> approach consistently outperforms the <italic>TPG</italic> baseline, reinforcing the effectiveness of embedding-based pair selection in enhancing the quality of learned representations.</p>
</sec>
<sec id="s5-2">
<title>5.2 Anomaly detection</title>
<p>Since the Tokyo dataset lacks annotations, we perform only a qualitative evaluation of anomaly detection performance. <xref ref-type="fig" rid="F5">Figure 5</xref> presents examples of anomalous events and indicates which methods were able to detect them. The shown events represent the four semantically distinct events with the highest anomaly score. These examples cover events with distinctive sounds, unique visual appearance, or a combination of both. This diversity demonstrates that the learned embedding space is semantically meaningful and contains information to identify various anomaly events.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Examples of anomalous events. Check marks indicate which models flagged this event as anomalous. <bold>(A)</bold> An advertising truck waits at the crossroad while an ambulance passes with its siren wailing. <bold>(B)</bold> A sports car speeds by with the engine roaring. <bold>(C)</bold> An ambulance enters from the upper left corner. <bold>(D)</bold> A forklift passes through without any distinct noise.</p>
</caption>
<graphic xlink:href="frobt-11-1490718-g005.tif"/>
</fig>
<p>In example (A), an advertising truck waits at the crossroad while an ambulance with wailing siren passes by. This anomaly is clearly identifiable through both visual and audio data. In example (B), a sports car speeds by with the engine roaring. While the sound of the sports car is highly distinctive, the car&#x2019;s visual appearance is not particularly notable. The Fusion baseline (<xref ref-type="bibr" rid="B26">Kumari and Saini, 2022</xref>), which relies on multimodal data, fails to flag this event as an anomaly. In example (C), an ambulance arrives from the topleft corner with sirens on, then turns right and exits in the bottom left corner. In this case, the ambulance is not visually prominent but is clearly audible. Example (D) shows a yellow forklift passing through the intersection. There is no distinctive engine sound, leading audio-only methods to miss this anomaly.</p>
<p>These experiments show that by mapping audio and visual to the same embedding space, we can learn representations that effectively integrate information from both modalities. This enhances the ability to detect anomalies that are challenging to identify using a single modality.</p>
</sec>
<sec id="s5-3">
<title>5.3 Event search</title>
<p>In our last set of experiments, we present examples of event search on the Tokyo dataset, as illustrated in <xref ref-type="fig" rid="F6">Figure 6</xref>. The top row shows still frames of the (manually selected) query clips, while the following rows showcase the most related events identified by each method, shown in increasing order of embedding distance. We excluded a 1-min window before and after each query event from being searched. Similarly, for each method, we show only search results that are at least 1 minute before or after higher ranked search results.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Examples of event query. The top row shows the query video, while the search results are shown in the 2&#x2013;6 rows. The results of each methods are shown in a decreasing order from left to right and top to down. <bold>(A)</bold> A pink bus. <bold>(B)</bold> An ambulance with sirens on. <bold>(C)</bold> A police car with broadcast and siren on. We show the top 4 search results obtained with our method for each query events. For the other three methods, only the top two search results are reported in order not to overload the figure.</p>
</caption>
<graphic xlink:href="frobt-11-1490718-g006.tif"/>
</fig>
<p>Example (A) shows a pink bus entering the scene while the sound of a passing train can be heard in the background, though the bus itself produces no distinctive sound. All methods successfully locate a similar event where the same pink bus appears at a different time, again accompanied by the sound of a train in the background. Notably, the overlap between the bus&#x2019;s visual presence and the train&#x2019;s audio is brief. While the other approaches prioritize frames with visual similarity over audio similarity, MAViL selects a frame emphasizing the distinct train sound. In contrast, our method shows only a 1% difference in preference between audio-similar and visual-similar frames. Moreover, our approach identifies an additional instance of the same bus later in the surveillance stream, this time without a train in the background. Other detections from all methods are quite diverse but typically contain either a bus, multiple black cars or the sound of a train in the background.</p>
<p>In example (B), an ambulance enters with its siren wailing accompanied by a broadcast announcement. Besides the passage in the query clip, the ambulance appears at least six more times during the night. All methods detect one particular instance where the ambulance enters from the bottom right corner and heads in the same direction as in the query clip. Interestingly, this event is detected earlier by <italic>EPG</italic> than by <italic>TPG</italic>, even before the ambulance was visible. This shows that even though <italic>EPG</italic> and <italic>TPG</italic> both integrate audio and video information, <italic>EPG</italic> is more adept at fusing both modalities, likely due to its training with embedding-based pair generation. By positively pairing segments where the ambulance is audible with segments where it is visible, <italic>EPG</italic> improves its ability to identify such events. Furthermore, in two other appearances the ambulance follows a different trajectory, entering from the top left and turning right. Only <italic>EPG</italic> and <italic>TPG</italic> successfully detect these cases. Additionally, there is one other instance where the broadcast audio is present, detected only by <italic>EPG</italic> and MAViL.</p>
<p>In example (C), a police car enters the scene with loud sirens and flashing warning lights. Although the car itself is rather small, its visual and auditory features make it distinct from other vehicles. <italic>EPG</italic>, <italic>TPG</italic>, and MVAiL all locate another similar event. Again, <italic>EPG</italic> detects the event earlier, even before the car is visible. All three methods also successfully identify other occurrences where police cars drive by or sirens are audible in the background. However, the baseline method (<italic>A &#x2b; V</italic>) is unable to find a matching event for this example. The examples of the ambulance and the police car demonstrate that our proposed embedding based pair generation and custom loss function help to keep more information and cover more aspects of the event, which benefits the real-world application.</p>
<p>Although this qualitative evaluation is limited in scope, it is important to emphasize that these results are derived using the same representations applied in the anomaly detection task. This highlights the versatility and robustness of the learned embeddings across multiple tasks.</p>
</sec>
<sec id="s5-4">
<title>5.4 Ablation study</title>
<p>To have a better understanding of our method, we analyze the results for different values of the hyperparameters used in our approach, namely, <inline-formula id="inf107">
<mml:math id="m112">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf108">
<mml:math id="m113">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Since <inline-formula id="inf109">
<mml:math id="m114">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> adapts based on the similarity of positive pairs, we focus our evaluation on <inline-formula id="inf110">
<mml:math id="m115">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf111">
<mml:math id="m116">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, which represent the temporal constraints and the weight between the distance functions, respectively. Since the only annotated task in our experiments is audio-visual event localization, we restrict our ablation study to this task.</p>
<sec id="s5-4-1">
<title>5.4.1 Temporal constraint <inline-formula id="inf112">
<mml:math id="m117">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</title>
<p>As shown in <xref ref-type="table" rid="T2">Table 2</xref>, the value of the temporal constraint <inline-formula id="inf113">
<mml:math id="m118">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> has minimal impact on our approach. Starting from the second epoch, the embedding-based pair mechanism is introduced, which does not solely rely on the time difference but also considers the semantic similarity in selecting pairs. This dual mechanism offers flexibility, as the optimal temporal constraint <inline-formula id="inf114">
<mml:math id="m119">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> can vary based on the content of the data. For instances, two frames that are 1 min apart in a scene depicting a sidewalk might contain very similar content, whereas the same time gap in a highway scene could result in significantly different content. Given this variability, the robustness of <italic>EPG</italic> shows another advantage in the real-world scenario.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Ablation study on different <inline-formula id="inf115">
<mml:math id="m120">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. The impact of different <inline-formula id="inf116">
<mml:math id="m121">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> on audio-visual event localization results.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">
<inline-formula id="inf117">
<mml:math id="m122">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>(sec)</th>
<th align="center">0</th>
<th align="center">1</th>
<th align="center">10</th>
<th align="center">30</th>
<th align="center">60</th>
<th align="center">120</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Similar</td>
<td align="center">86.92</td>
<td align="center">86.21</td>
<td align="center">85.32</td>
<td align="center">87.41</td>
<td align="center">87.92</td>
<td align="center">86.31</td>
</tr>
<tr>
<td align="left">Challenging</td>
<td align="center">77.48</td>
<td align="center">78.62</td>
<td align="center">77.33</td>
<td align="center">79.82</td>
<td align="center">79.82</td>
<td align="center">78.32</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s5-4-2">
<title>5.4.2 Weight in distance function <inline-formula id="inf118">
<mml:math id="m123">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</title>
<p>
<xref ref-type="table" rid="T3">Table 3</xref> shows the effect of different values for <inline-formula id="inf119">
<mml:math id="m124">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, which is used in <xref ref-type="disp-formula" rid="e5">Equation 5</xref>. Both cosine similarity and Euclidean distance are commonly used as distance metrics in contrastive learning (<xref ref-type="bibr" rid="B20">Hadsell et al., 2006</xref>; <xref ref-type="bibr" rid="B56">Zeng et al., 2020</xref>). When <inline-formula id="inf120">
<mml:math id="m125">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is set to 0.25, 0.5, or 0.75, there is no significant impact on the performance of audio-visual event localization. However, when only one of the distance metrics is used <inline-formula id="inf121">
<mml:math id="m126">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, we observe that the model sometimes fails to converge. We investigated the learning process and hypothesize why the instability arises when using only Euclidean distance or cosine similarity. On the one hand, as the pre-trained visual encoder <inline-formula id="inf122">
<mml:math id="m127">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is not normalized to a unit vector, we did not normalize the output of <inline-formula id="inf123">
<mml:math id="m128">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> either. In the early stages of the training, using only cosine similarity occasionally leads to the model collapsing or diverging. Cosine similarity does not reflect the magnitude of the vector, which can be a crucial factor in measuring the spatial correlation between two data points in a non-unified embedding space.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Ablation study on different <inline-formula id="inf124">
<mml:math id="m129">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. DNC stands for <italic>Did not converge</italic>. The impact of different <inline-formula id="inf125">
<mml:math id="m130">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> on audio-visual event localization results.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">
<inline-formula id="inf126">
<mml:math id="m131">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th align="center">0</th>
<th align="center">0.25</th>
<th align="center">0.5</th>
<th align="center">0.75</th>
<th align="center">1</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Similar</td>
<td align="center">DNC</td>
<td align="center">85.21</td>
<td align="center">86.92</td>
<td align="center">85.44</td>
<td align="center">DNC</td>
</tr>
<tr>
<td align="left">Challenging</td>
<td align="center">DNC</td>
<td align="center">75.16</td>
<td align="center">77.48</td>
<td align="center">76.83</td>
<td align="center">DNC</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In contrast, Euclidean distance provides more efficient guidance in training the <inline-formula id="inf127">
<mml:math id="m132">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> more efficiently. However, we found that a poor choice of the margin constant also leads to model collapse or divergence when using only Euclidean distance. Since the optimal margin for Euclidean distance can vary depending on the data, incorporating cosine similarity helps balance the loss function, leading to more stable training. Moreover, Euclidean distance complements cosine similarity by considering the absolute difference between two vectors. The combination of both metrics helps in identifying semantically similar events in the embedding space, as it captures both directional and magnitude-based relationships.</p>
</sec>
</sec>
</sec>
<sec id="s6">
<title>6 Conclusion and future work</title>
<p>In this paper, we discussed the challenges of learning audio-visual representations from multi-modal surveillance data. We addressed the limitations of relying solely on temporal alignment as pretext task, as well as the minimal sufficient representation bottleneck inherent in contrastive learning. To the best of our knowledge, this is the first study to explore these issues in the context of surveillance data.</p>
<p>We introduced a novel embedding-based pair generation mechanism that mitigates the problem of false negative pair generation while promoting more diversity in positive pairs. Our pseudo-Siamese network, enhanced by a new contrastive loss function that accounts for multiple positive pairs, learns more effective audio-visual representations.</p>
<p>We evaluated the generalization of the learned representations across various downstream tasks and compared our approach to state-of-the-art approaches using a publicly available dataset. Additionally, we demonstrated the effectiveness of our method on a more challenging dataset of real-world surveillance data. Our results show that our approach performs similar or better than existing state-of-the-art techniques.</p>
<p>In future work, we will further explore how the learned representations perform on different downstream tasks. We will also investigate techniques to make the model smaller and faster. Both audio and video data are potentially privacy sensitive and should not leave the local edge device unless absolutely necessary. To facilitate this, we aim to optimize the model for real-time operation on resource-constrained platforms, enabling scalable deployment while preserving privacy.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found in the article/supplementary material.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>W-CW: Conceptualization, Formal Analysis, Investigation, Methodology, Software, Writing&#x2013;original draft, Writing&#x2013;review and editing. SD: Writing&#x2013;review and editing. SL: Conceptualization, Supervision, Writing&#x2013;review and editing. PS: Conceptualization, Supervision, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. Flemish Government under the &#x201c;Onderzoeksprogramma Artifici&#xeb;le Intelligentie (AI) Vlaanderen&#x201d; programme.</p>
</sec>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn id="fn1">
<label>1</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://github.com/sainathadapa/dcase2019-task5-urban-sound-tagging">https://github.com/sainathadapa/dcase2019-task5-urban-sound-tagging</ext-link>
</p>
</fn>
<fn id="fn2">
<label>2</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://www.youtube.com/watch?v=2gZySUir8_w$">https://www.youtube.com/watch?v&#x3d;2gZySUir8_w$</ext-link>
</p>
</fn>
<fn id="fn3">
<label>3</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://www.youtube.com/watch?v=xiLF6PmFZP4$">https://www.youtube.com/watch?v&#x3d;xiLF6PmFZP4$</ext-link>
</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Adapa</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Urban sound tagging using convolutional neural networks</article-title>. <comment>
<italic>arXiv preprint arXiv:1909.12699</italic>
</comment>
</citation>
</ref>
<ref id="B2">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Afouras</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chung</surname>
<given-names>J. S.</given-names>
</name>
<name>
<surname>Zisserman</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020a</year>). &#x201c;<article-title>Asr is all you need: cross-modal distillation for lip reading</article-title>,&#x201d; in <source>ICASSP 2020-2020 IEEE international conference on acoustics, speech and signal processing (ICASSP)</source> (<publisher-name>IEEE</publisher-name>), <fpage>2143</fpage>&#x2013;<lpage>2147</lpage>.</citation>
</ref>
<ref id="B3">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Afouras</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Owens</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chung</surname>
<given-names>J. S.</given-names>
</name>
<name>
<surname>Zisserman</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020b</year>). &#x201c;<article-title>Self-supervised learning of audio-visual objects from video</article-title>,&#x201d; in <source>Computer vision&#x2013;ECCV 2020: 16th European conference, glasgow, UK, august 23&#x2013;28, 2020, proceedings, Part XVIII 16</source> (<publisher-name>Springer</publisher-name>), <fpage>208</fpage>&#x2013;<lpage>224</lpage>.</citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Arandjelovic</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zisserman</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Look, listen and learn</article-title>,&#x201d; in <source>Proceedings of the IEEE international conference on computer vision</source>, <fpage>609</fpage>&#x2013;<lpage>617</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aytar</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Vondrick</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Torralba</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Soundnet: learning sound representations from unlabeled video</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>29</volume>. <pub-id pub-id-type="doi">10.5555/3157096.3157196</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bai</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Multimodal urban sound tagging with spatiotemporal context</article-title>. <source>IEEE Trans. Cognitive Dev. Syst.</source> <volume>15</volume>, <fpage>555</fpage>&#x2013;<lpage>565</lpage>. <pub-id pub-id-type="doi">10.1109/tcds.2022.3160168</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Bajovic</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Bakhtiarnia</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bravos</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Brutti</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Burkhardt</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Cauchi</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). &#x201c;<article-title>Marvel: multimodal extreme scale data analytics for smart cities environments</article-title>,&#x201d; in <source>2021 international balkan conference on communications and networking (BalkanCom)</source> (<publisher-name>IEEE</publisher-name>), <fpage>143</fpage>&#x2013;<lpage>147</lpage>.</citation>
</ref>
<ref id="B8">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Belilovsky</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Eickenberg</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Oyallon</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Greedy layerwise learning can scale to imagenet</article-title>,&#x201d; in <source>International conference on machine learning</source> (<publisher-loc>Brookline, Massachusetts, USA</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>583</fpage>&#x2013;<lpage>593</lpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Benfold</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Reid</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Stable multi-target tracking in real-time surveillance video</article-title>. <source>CVPR</source> <volume>2011</volume>, <fpage>3457</fpage>&#x2013;<lpage>3464</lpage>. <pub-id pub-id-type="doi">10.1109/cvpr.2011.5995667</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Boudiaf</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Rony</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ziko</surname>
<given-names>I. M.</given-names>
</name>
<name>
<surname>Granger</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Pedersoli</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Piantanida</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). &#x201c;<article-title>A unifying mutual information view of metric learning: cross-entropy vs. pairwise losses</article-title>,&#x201d; in <source>European conference on computer vision</source> (<publisher-name>Springer</publisher-name>), <fpage>548</fpage>&#x2013;<lpage>564</lpage>.</citation>
</ref>
<ref id="B11">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kornblith</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Norouzi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>A simple framework for contrastive learning of visual representations</article-title>,&#x201d; in <source>International conference on machine learning</source> (<publisher-loc>Brookline, Massachusetts, USA</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>1597</fpage>&#x2013;<lpage>1607</lpage>.</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chuang</surname>
<given-names>C.-Y.</given-names>
</name>
<name>
<surname>Robinson</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y.-C.</given-names>
</name>
<name>
<surname>Torralba</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Jegelka</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Debiased contrastive learning</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>33</volume>, <fpage>8765</fpage>&#x2013;<lpage>8775</lpage>. <pub-id pub-id-type="doi">10.5555/3495724.3496459</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Danesh Pazho</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Alinezhad Noghre</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Rahimi Ardabili</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Neff</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Tabkhi</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Chad: charlotte anomaly dataset</article-title>,&#x201d; in <source>Image analysis</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer Nature Switzerland</publisher-name>), <fpage>50</fpage>&#x2013;<lpage>66</lpage>.</citation>
</ref>
<ref id="B15">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Feichtenhofer</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>X3d: expanding architectures for efficient video recognition</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>203</fpage>&#x2013;<lpage>213</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Feichtenhofer</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Girshick</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>A large-scale study on unsupervised spatiotemporal representation learning</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>3299</fpage>&#x2013;<lpage>3309</lpage>.</citation>
</ref>
<ref id="B17">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Feng</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Owens</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Self-supervised video forensics by audio-visual anomaly detection</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>10491</fpage>&#x2013;<lpage>10503</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gong</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Audio&#x2013;visual representation learning for anomaly events detection in crowds</article-title>. <source>Neurocomputing</source> <volume>582</volume>, <fpage>127489</fpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2024.127489</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Gidaris</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Komodakis</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Unsupervised representation learning by predicting image rotations</article-title>,&#x201d; in <source>International conference on learning representations</source>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hadsell</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Chopra</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>LeCun</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Dimensionality reduction by learning an invariant mapping</article-title>. <source>2006 IEEE Comput. Soc. Conf. Comput. Vis. pattern Recognit. (CVPR&#x2019;06) (IEEE)</source> <volume>2</volume>, <fpage>1735</fpage>&#x2013;<lpage>1742</lpage>. <pub-id pub-id-type="doi">10.1109/cvpr.2006.100</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Doll&#xe1;r</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Girshick</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Masked autoencoders are scalable vision learners</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>16000</fpage>&#x2013;<lpage>16009</lpage>.</citation>
</ref>
<ref id="B22">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Girshick</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Momentum contrast for unsupervised visual representation learning</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>9729</fpage>&#x2013;<lpage>9738</lpage>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>P.-Y.</given-names>
</name>
<name>
<surname>Sharma</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ryali</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.-W.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Mavil: masked audio-video learners</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>36</volume>. <pub-id pub-id-type="doi">10.5555/3666122.3667016</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kalantidis</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sariyildiz</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>Pion</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Weinzaepfel</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Larlus</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Hard negative mixing for contrastive learning</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>33</volume>, <fpage>21798</fpage>&#x2013;<lpage>21809</lpage>. <pub-id pub-id-type="doi">10.5555/3495724.3497553</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khosla</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Teterwak</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sarna</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Isola</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Supervised contrastive learning</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>33</volume>, <fpage>18661</fpage>&#x2013;<lpage>18673</lpage>. <pub-id pub-id-type="doi">10.5555/3495724.3497291</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kumari</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Saini</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>An adaptive framework for anomaly detection in time-series audio-visual data</article-title>. <source>IEEE Access</source> <volume>10</volume>, <fpage>36188</fpage>&#x2013;<lpage>36199</lpage>. <pub-id pub-id-type="doi">10.1109/access.2022.3164439</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Leporowski</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Bakhtiarnia</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bonnici</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Muscat</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zanella</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Audio-visual dataset and method for anomaly detection in traffic videos</article-title>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Leroux</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Simoens</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Automated training of location-specific edge models for traffic counting</article-title>. <source>Comput. Electr. Eng.</source> <volume>99</volume>, <fpage>107763</fpage>. <pub-id pub-id-type="doi">10.1016/j.compeleceng.2022.107763</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Lian</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Future frame prediction for anomaly detection--a new baseline</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>, <fpage>6536</fpage>&#x2013;<lpage>6545</lpage>.</citation>
</ref>
<ref id="B30">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2013</year>). &#x201c;<article-title>Abnormal event detection at 150 fps in matlab</article-title>,&#x201d; in <source>Proceedings of the IEEE international conference on computer vision</source>, <fpage>2720</fpage>&#x2013;<lpage>2727</lpage>.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Malon</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Roman-Jimenez</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Guyot</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Chambon</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Charvillat</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Crouzil</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Toulouse Campus Surveillance Dataset: scenarios, soundtracks, synchronized videos with overlapping and disjoint views</article-title>. <pub-id pub-id-type="doi">10.5281/zenodo.3697806</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mao</surname>
<given-names>Q.-C.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>H.-M.</given-names>
</name>
<name>
<surname>Zuo</surname>
<given-names>L.-Q.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>R.-S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Finding every car: a traffic surveillance multi-scale vehicle object detection method</article-title>. <source>Appl. Intell.</source> <volume>50</volume>, <fpage>3125</fpage>&#x2013;<lpage>3136</lpage>. <pub-id pub-id-type="doi">10.1007/s10489-020-01704-5</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Misra</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Maaten</surname>
<given-names>L. v. d.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Self-supervised learning of pretext-invariant representations</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>6707</fpage>&#x2013;<lpage>6717</lpage>.</citation>
</ref>
<ref id="B34">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Morgado</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Misra</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Vasconcelos</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Robust audio-visual instance discrimination</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>12934</fpage>&#x2013;<lpage>12945</lpage>.</citation>
</ref>
<ref id="B35">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Munjal</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Amin</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tombari</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Galasso</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Query-guided end-to-end person search</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>811</fpage>&#x2013;<lpage>820</lpage>.</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mydlarz</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Salamon</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Bello</surname>
<given-names>J. P.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>The implementation of low-cost urban acoustic monitoring devices</article-title>. <source>Appl. Acoust.</source> <volume>117</volume>, <fpage>207</fpage>&#x2013;<lpage>218</lpage>. <pub-id pub-id-type="doi">10.1016/j.apacoust.2016.06.010</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Perez</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kot</surname>
<given-names>A. C.</given-names>
</name>
<name>
<surname>Rocha</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Detection of real-world fights in surveillance videos</article-title>,&#x201d; in <source>Icassp 2019 - 2019 IEEE international conference on acoustics, speech and signal processing (ICASSP)</source>, <fpage>2662</fpage>&#x2013;<lpage>2666</lpage>. <pub-id pub-id-type="doi">10.1109/ICASSP.2019.8683676</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Qian</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Gong</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>M.-H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Belongie</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). &#x201c;<article-title>Spatiotemporal contrastive video representation learning</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>6964</fpage>&#x2013;<lpage>6974</lpage>.</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ran</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Self-supervised video representation and temporally adaptive attention for audio-visual event localization</article-title>. <source>Appl. Sci.</source> <volume>12</volume>, <fpage>12622</fpage>. <pub-id pub-id-type="doi">10.3390/app122412622</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ristani</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Solera</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>R. S.</given-names>
</name>
<name>
<surname>Cucchiara</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Tomasi</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Performance measures and a data set for multi-target, multi-camera tracking</article-title>,&#x201d; in <source>ECCV workshops</source>.</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sampath</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Maurtua</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Aguilar Martin</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Gutierrez</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A survey on generative adversarial networks for imbalance problems in computer vision tasks</article-title>. <source>J. big Data</source> <volume>8</volume>, <fpage>27</fpage>&#x2013;<lpage>59</lpage>. <pub-id pub-id-type="doi">10.1186/s40537-021-00414-0</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shah</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sra</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chellappa</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Cherian</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Max-margin contrastive learning</article-title>. <source>Proc. AAAI Conf. Artif. Intell.</source> <volume>36</volume>, <fpage>8220</fpage>&#x2013;<lpage>8230</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v36i8.20796</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). &#x201c;<article-title>Learning audio-visual source localization via false negative aware contrastive learning</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>6420</fpage>&#x2013;<lpage>6429</lpage>.</citation>
</ref>
<ref id="B44">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Efficientnet: rethinking model scaling for convolutional neural networks</article-title>,&#x201d; in <source>International conference on machine learning</source> (<publisher-loc>Brookline, Massachusetts, USA</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>6105</fpage>&#x2013;<lpage>6114</lpage>.</citation>
</ref>
<ref id="B45">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tian</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Krishnan</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Isola</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2020a</year>). &#x201c;<article-title>Contrastive multiview coding</article-title>,&#x201d; in <source>Computer vision&#x2013;ECCV 2020: 16th European conference, glasgow, UK, august 23&#x2013;28, 2020, proceedings, Part XI 16</source> (<publisher-name>Springer</publisher-name>), <fpage>776</fpage>&#x2013;<lpage>794</lpage>.</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tian</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Poole</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Krishnan</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Schmid</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Isola</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2020b</year>). <article-title>What makes for good views for contrastive learning?</article-title> <source>Adv. neural Inf. Process. Syst.</source> <volume>33</volume>, <fpage>6827</fpage>&#x2013;<lpage>6839</lpage>. <pub-id pub-id-type="doi">10.5555/3495724.3496297</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tsai</surname>
<given-names>Y.-H.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Salakhutdinov</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Morency</surname>
<given-names>L.-P.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Self-supervised learning from a multi-view perspective</article-title>,&#x201d; in <source>Proceedings of the international conference on learning representations</source> (<publisher-name>ICLR</publisher-name>).</citation>
</ref>
<ref id="B48">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tseng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Berry</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.-T.</given-names>
</name>
<name>
<surname>Chiu</surname>
<given-names>I.-H.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>H.-H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). &#x201c;<article-title>Av-superb: a multi-task evaluation benchmark for audio-visual representation models</article-title>,&#x201d; in <source>ICASSP 2024-2024 IEEE international conference on acoustics, speech and signal processing (ICASSP)</source> (<publisher-name>IEEE</publisher-name>), <fpage>6890</fpage>&#x2013;<lpage>6894</lpage>.</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ullah</surname>
<given-names>F. U. M.</given-names>
</name>
<name>
<surname>Obaidat</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Ullah</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Muhammad</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hijji</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Baik</surname>
<given-names>S. W.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A comprehensive review on vision-based violence detection in surveillance videos</article-title>. <source>ACM Comput. Surv.</source> <volume>55</volume>, <fpage>1</fpage>&#x2013;<lpage>44</lpage>. <pub-id pub-id-type="doi">10.1145/3561971</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="web">
<collab>United Nations Department of Economic and Social Affairs</collab>(<year>2018</year>). <article-title>World urbanization prospects: The 2018 revision</article-title>. <source>Statistical Papers - United Nations (Ser. A), Population and Vital Statistics Report</source>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://www.un-ilibrary.org/content/books/9789210043144">https://www.un-ilibrary.org/content/books/9789210043144</ext-link>
</comment>. <pub-id pub-id-type="doi">10.18356/b9e995fe-en</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>Z.-H.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Rethinking minimal sufficient representation in contrastive learning</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>16041</fpage>&#x2013;<lpage>16050</lpage>.</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Albrecht</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Braham</surname>
<given-names>N. A. A.</given-names>
</name>
<name>
<surname>Mou</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>X. X.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Self-supervised learning in remote sensing: A review</article-title>,&#x201d; in <source>IEEE Geoscience and Remote Sensing Magazine</source> <volume>10</volume>, <fpage>213</fpage>&#x2013;<lpage>247</lpage>. <pub-id pub-id-type="doi">10.1109/MGRS.2022.3198244</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>J.-C.</given-names>
</name>
<name>
<surname>Hsieh</surname>
<given-names>H.-Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>D.-J.</given-names>
</name>
<name>
<surname>Fuh</surname>
<given-names>C.-S.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>T.-L.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Self-supervised sparse representation for video anomaly detection</article-title>,&#x201d; in <source>European conference on computer vision</source> (<publisher-name>Springer</publisher-name>), <fpage>729</fpage>&#x2013;<lpage>745</lpage>.</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Dss-net: dynamic self-supervised network for video anomaly detection</article-title>. <source>IEEE Trans. Multimedia</source> <volume>26</volume>, <fpage>2124</fpage>&#x2013;<lpage>2136</lpage>. <pub-id pub-id-type="doi">10.1109/tmm.2023.3292596</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wuerkaixi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Duan</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Rethinking audio-visual synchronization for active speaker detection</article-title>,&#x201d; in <source>
<italic>2022 IEEE 32nd international Workshop on machine Learning for signal processing (MLSP)</italic> (IEEE)</source>, <fpage>01</fpage>&#x2013;<lpage>06</lpage>.</citation>
</ref>
<ref id="B55">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zbontar</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jing</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Misra</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>LeCun</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Deny</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Barlow twins: self-supervised learning via redundancy reduction</article-title>,&#x201d; in <source>International conference on machine learning</source> (<publisher-loc>Brookline, Massachusetts, USA</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>12310</fpage>&#x2013;<lpage>12320</lpage>.</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zeng</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Oyama</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Deep triplet neural networks with cluster-cca for audio-visual cross-modal retrieval</article-title>. <source>ACM Trans. Multimedia Comput. Commun. Appl. (TOMM)</source> <volume>16</volume>, <fpage>1</fpage>&#x2013;<lpage>23</lpage>. <pub-id pub-id-type="doi">10.1145/3387164</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>On the generalization of multi-modal contrastive learning</article-title>,&#x201d; in <source>Proceedings of the 40th international conference on machine learning</source> (<publisher-loc>Brookline, Massachusetts, USA</publisher-loc>: <publisher-name>JMLR</publisher-name>). <comment>ICML&#x2019;23</comment>.</citation>
</ref>
<ref id="B58">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ouyang</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Exploiting visual context semantics for sound source localization</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF winter conference on applications of computer vision</source>, <fpage>5199</fpage>&#x2013;<lpage>5208</lpage>.</citation>
</ref>
<ref id="B59">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>C. W.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Improving contrastive learning by visualizing feature transformation</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF international conference on computer vision</source>, <fpage>10306</fpage>&#x2013;<lpage>10315</lpage>.</citation>
</ref>
<ref id="B60">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zolfaghari</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gehler</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Brox</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Crossclr: cross-modal contrastive learning for multi-modal video representations</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF international conference on computer vision</source>, <fpage>1450</fpage>&#x2013;<lpage>1459</lpage>.</citation>
</ref>
</ref-list>
</back>
</article>