@inproceedings{0ac36c7c2a31404d8bb29974d5f0e3a4,
title = "Attributes and action recognition based on convolutional neural networks and spatial pyramid VLAD encoding",
abstract = "Determination of human attributes and recognition of actions in still images are two related and challenging tasks in computer vision, which often appear in fine-grained domains where the distinctions between the different categories are very small. Deep Convolutional Neural Network (CNN) models have demonstrated their remarkable representational learning capability through various examples. However, the successes are very limited for attributes and action recognition as the potential of CNNs to acquire both of the global and local information of an image remains largely unexplored. This paper proposes to tackle the problem with an encoding of a spatial pyramid Vector of Locally Aggregated Descriptors (VLAD) on top of CNN features. With region proposals generated by Edgeboxes, a compact and efficient representation of an image is thus produced for subsequent prediction of attributes and classification of actions. The proposed scheme is validated with competitive results on two benchmark datasets: 90.4% mean Average Precision (mAP) on the Berkeley Attributes of People dataset and 88.5% mAP on the Stanford 40 action dataset.",
author = "Shiyang Yan and Smith, {Jeremy S.} and Bailing Zhang",
note = "Publisher Copyright: {\textcopyright} Springer International Publishing AG 2017.; 13th Asian Conference on Computer Vision, ACCV 2016 ; Conference date: 20-11-2016 Through 24-11-2016",
year = "2017",
doi = "10.1007/978-3-319-54526-4_37",
language = "English",
isbn = "9783319545257",
series = "Lecture Notes in Computer Science (including subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics)",
publisher = "Springer Verlag",
pages = "500--514",
editor = "Chu-Song Chen and Kai-Kuang Ma and Jiwen Lu",
booktitle = "Computer Vision - ACCV 2016 Workshops, ACCV 2016 International Workshops, Revised Selected Papers",
}