@inproceedings{e2ada864cb6f4877a924dcd6affdff35,
title = "Span-based Audio-Visual Localization",
abstract = "This paper focuses on the audio-visual event localization task that aims to match both visible and audible components in a video to identify the event of interest. Existing methods primarily ignore the continuity of audio-visual events and classify each segment separately. They either classify the event category score of each segment separately or calculate the event-relevant score of each segment separately. However, events in video are often continuous and last several segments. Motivated by these, we propose a span-based framework that considers consecutive segments jointly. The span-based framework handles the audio-visual localization task by predicting the event class and extracting the event span. Specifically, a [CLS] token is applied to collect the global information with self-attention mechanisms to predict the event class. Relevance scores and positional embeddings are inserted into the span predictor to estimate the start and end boundaries of the event. Multi-modal Mixup are further used to improve the robustness and generalization of the model. Experiments conducted on the AVE dataset demonstrate that the proposed method outperforms state-of-the-art methods.",
keywords = "audio-visual localization, multi-modal, span prediction",
author = "Yiling Wu and Xinfeng Zhang and Yaowei Wang and Qingming Huang",
note = "Publisher Copyright: {\textcopyright} 2022 ACM.; 30th ACM International Conference on Multimedia, MM 2022 ; Conference date: 10-10-2022 Through 14-10-2022",
year = "2022",
month = oct,
day = "10",
doi = "10.1145/3503161.3548318",
language = "英语",
series = "MM 2022 - Proceedings of the 30th ACM International Conference on Multimedia",
publisher = "Association for Computing Machinery, Inc",
pages = "1252--1260",
booktitle = "MM 2022 - Proceedings of the 30th ACM International Conference on Multimedia",
}