@inproceedings{89832c9030b94d018fab1e523ab01e9d,
title = "Recognizing Conversational Interaction Based on 3D Human Pose",
abstract = "In this paper, we take a bag of visual words approach to investigate whether it is possible to distinguish conversational scenarios from observing human motion alone, in particular gestures in 3D. The conversational interactions concerned in this work have rather subtle differences among them. Unlike typical action or event recognition, each interaction in our case contain many instances of primitive motions and actions, many of which are shared among different conversation scenarios. Hence, extracting and learning temporal dynamics are essential. We adopt Kinect sensors to extract low level temporal features. These features are then generalized to form a visual vocabulary that can be further generalized to a set of topics from temporal distributions of visual vocabulary. A subject-specific supervised learning approach based on both generative and discriminative classifiers is employed to classify the testing sequences to seven different conversational scenarios. We believe this is among one of the first work that is devoted to conversational interaction classification using 3D pose features and to show this task is indeed possible.",
keywords = "3D human pose, conversational interaction classification, interaction analysis, Kinect sensor, MOTION CAPTURE, RECOGNITION, REPRESENTATION",
author = "Jingjing Deng and Xianghua Xie and Ben Daubney and Hui Fang and Grant, {Phil W.}",
year = "2013",
doi = "10.1007/978-3-319-02895-8_13",
language = "English",
isbn = "978-3-319-02894-1",
series = "Lecture Notes in Computer Science",
publisher = "Springer",
pages = "138--149",
editor = "J BlancTalon and A Kasinski and W Philips and D Popescu and P Scheunders",
booktitle = "Advanced Concepts for Intelligent Vision Systems",
address = "Germany",
note = "15th International Conference on Advanced Concepts for Intelligent Vision Systems (ACIVS) ; Conference date: 28-10-2013 Through 31-10-2013",
}