@inproceedings{3fd5462922e94eb9b2cf67fc9114fd36,
title = "Select and focus: Action recognition with spatial-temporal attention",
abstract = "With the rapid development of neural networks, human action recognition has been achieved great improvement by using convolutional neural networks (CNN) or recurrent neural networks (RNN). In this paper, we propose a model based on weighted spatial-temporal attention for action recognition. This model selects the key parts in each video frame and important frames in each video sequence. Then the model focuses on analyzing these key parts and frames. Therefore, the most important tasks of our model is to find out the key parts spatially and the important frames temporally for recognizing the action. Our model is trained and tested on three datasets including UCF-11, UCF-101, and HMDB51. The experiments demonstrate that our model can achieve a satisfactory result for human action recognition.",
keywords = "Attention, Deep learning, Human action recognition",
author = "Wensong Chan and Zhiqiang Tian and Shuai Liu and Jing Ren and Xuguang Lan",
note = "Publisher Copyright: {\textcopyright} 2019, Springer Nature Switzerland AG.; 12th International Conference on Intelligent Robotics and Applications, ICIRA 2019 ; Conference date: 08-08-2019 Through 11-08-2019",
year = "2019",
doi = "10.1007/978-3-030-27535-8\_41",
language = "英语",
isbn = "9783030275341",
series = "Lecture Notes in Computer Science (including subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics)",
publisher = "Springer Verlag",
pages = "461--471",
editor = "Haibin Yu and Jinguo Liu and Lianqing Liu and Yuwang Liu and Zhaojie Ju and Dalin Zhou",
booktitle = "Intelligent Robotics and Applications - 12th International Conference, ICIRA 2019, Proceedings",
}