{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/multi-region-two-stream-r-cnn-for-action","title":"Multi-region two-stream R-CNN for action detection","arxiv_id":null,"date":"2016-09-17","proceeding":"European Conference on Computer Vision (ECVV 2016) 2016 9","authors":["Xiaojiang Peng","Cordelia Schmid"],"abstract":"We propose a multi-region two-stream R-CNN model for action detection in realistic videos. We start from frame-level action detection based on faster R-CNN [1], and make three contributions: (1) we show that a motion region proposal network generates high-quality proposals , which are complementary to those of an appearance region proposal network; (2) we show that stacking optical flow over several frames significantly improves frame-level action detection; and (3) we embed a multi-region scheme in the faster R-CNN model, which adds complementary information on body parts. We then link frame-level detections with the Viterbi algorithm, and temporally localize an action with the maximum subarray method. Experimental results on the UCF-Sports, J-HMDB and UCF101 action detection datasets show that our approach outperforms the state of the art with a significant margin in both frame-mAP and video-mAP","url_abs":"https://doi.org/10.1007/978-3-319-46493-0_45","url_pdf":"https://hal.inria.fr/hal-01349107v1/document","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[],"tasks":[{"task_slug":"action-detection","task_name":"Action Detection"},{"task_slug":"action-recognition-in-videos","task_name":"Action Recognition"},{"task_slug":"region-proposal","task_name":"Region Proposal"},{"task_slug":"skeleton-based-action-recognition","task_name":"Skeleton Based Action Recognition"}],"methods":[{"method_slug":"convolution","method_name":"Convolution"},{"method_slug":"faster-r-cnn","method_name":"Faster R-CNN"},{"method_slug":"rpn","method_name":"RPN"},{"method_slug":"roipool","method_name":"RoIPool"},{"method_slug":"softmax","method_name":"Softmax"}],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/action-detection-on-j-hmdb","task":"Action Detection","dataset":"J-HMDB","model":"MR-TS R-CNN","rank_in_archive_order":10,"of":18,"metrics":{"Frame-mAP 0.5":"58.5","Video-mAP 0.2":"74.3","Video-mAP 0.5":"73.09"},"uses_additional_data":false},{"leaderboard":"/sota/action-detection-on-j-hmdb","task":"Action Detection","dataset":"J-HMDB","model":"TS R-CNN","rank_in_archive_order":11,"of":18,"metrics":{"Frame-mAP 0.5":"56.9","Video-mAP 0.2":"71.1","Video-mAP 0.5":"70.6"},"uses_additional_data":false},{"leaderboard":"/sota/action-detection-on-ucf-sports","task":"Action Detection","dataset":"UCF Sports","model":"MR-TS R-CNN","rank_in_archive_order":2,"of":7,"metrics":{"Frame-mAP 0.5":"84.52","Video-mAP 0.2":"94.83","Video-mAP 0.5":"94.67"},"uses_additional_data":false},{"leaderboard":"/sota/action-detection-on-ucf-sports","task":"Action Detection","dataset":"UCF Sports","model":"TS R-CNN","rank_in_archive_order":3,"of":7,"metrics":{"Frame-mAP 0.5":"82.30","Video-mAP 0.2":"94.82","Video-mAP 0.5":"94.82"},"uses_additional_data":false},{"leaderboard":"/sota/action-detection-on-ucf101-24","task":"Action Detection","dataset":"UCF101-24","model":"TS R-CNN","rank_in_archive_order":14,"of":19,"metrics":{"Frame-mAP 0.5":"39.94"},"uses_additional_data":false},{"leaderboard":"/sota/action-detection-on-ucf101-24","task":"Action Detection","dataset":"UCF101-24","model":"MR-TS R-CNN","rank_in_archive_order":15,"of":19,"metrics":{"Frame-mAP 0.5":"39.63"},"uses_additional_data":false},{"leaderboard":"/sota/action-recognition-in-videos-on-ucf101","task":"Action Recognition","dataset":"UCF101","model":"MR Two-Sream R-CNN","rank_in_archive_order":71,"of":91,"metrics":{"3-fold Accuracy":"91.1"},"uses_additional_data":false},{"leaderboard":"/sota/skeleton-based-action-recognition-on-j-hmdb","task":"Skeleton Based Action Recognition","dataset":"J-HMDB","model":"MR Two-Sream R-CNN","rank_in_archive_order":7,"of":13,"metrics":{"Accuracy (RGB+pose)":"71.1"},"uses_additional_data":false}],"syntology":{"atlas_url":null,"mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}