@inproceedings{a40cd61d4a9e40f988fce2a2afb37fb0,
title = "A Slow-I-Fast-P Architecture for Compressed Video Action Recognition",
abstract = "Compressed video action recognition has drawn growing attention for the storage and processing advantages of compressed videos over original raw videos. While the past few years have witnessed remarkable progress in this problem, most existing approaches rely on RGB frames from raw videos and require multi-step training. In this paper, we propose a novel Slow-I-Fast-P (SIFP) neural network model for compressed video action recognition. It consists of the slow I pathway receiving a sparse sampling I-frame clip and the fast P pathway receiving a dense sampling pseudo optical flow clip. An unsupervised estimation method and a new loss function are designed to generate pseudo optical flows in compressed videos. Our model eliminates the dependence on the traditional optical flows calculated from raw videos. The model is trained in an end-to-end way. The proposed method is evaluated on the challenging HMDB51 and UCF101 datasets. The extensive comparison results and ablation studies demonstrate the effectiveness and strength of the proposed method.",
keywords = "action recognition, compressed video, neural networks",
author = "Jiapeng Li and Ping Wei and Yongchi Zhang and Nanning Zheng",
note = "Publisher Copyright: {\textcopyright} 2020 Owner/Author.; 28th ACM International Conference on Multimedia, MM 2020 ; Conference date: 12-10-2020 Through 16-10-2020",
year = "2020",
month = oct,
day = "12",
doi = "10.1145/3394171.3413641",
language = "英语",
series = "MM 2020 - Proceedings of the 28th ACM International Conference on Multimedia",
publisher = "Association for Computing Machinery, Inc",
pages = "2039--2047",
booktitle = "MM 2020 - Proceedings of the 28th ACM International Conference on Multimedia",
}