{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/two-stream-convolutional-networks-for-action","title":"Two-Stream Convolutional Networks for Action Recognition in Videos","arxiv_id":"1406.2199","date":"2014-06-09","proceeding":"NeurIPS 2014 12","authors":["Karen Simonyan","Andrew Zisserman"],"abstract":"We investigate architectures of discriminatively trained deep Convolutional\nNetworks (ConvNets) for action recognition in video. The challenge is to\ncapture the complementary information on appearance from still frames and\nmotion between frames. We also aim to generalise the best performing\nhand-crafted features within a data-driven learning framework.\n  Our contribution is three-fold. First, we propose a two-stream ConvNet\narchitecture which incorporates spatial and temporal networks. Second, we\ndemonstrate that a ConvNet trained on multi-frame dense optical flow is able to\nachieve very good performance in spite of limited training data. Finally, we\nshow that multi-task learning, applied to two different action classification\ndatasets, can be used to increase the amount of training data and improve the\nperformance on both.\n  Our architecture is trained and evaluated on the standard video actions\nbenchmarks of UCF-101 and HMDB-51, where it is competitive with the state of\nthe art. It also exceeds by a large margin previous attempts to use deep nets\nfor video classification.","url_abs":"http://arxiv.org/abs/1406.2199v2","url_pdf":"http://arxiv.org/pdf/1406.2199v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"two-stream-convolutional-networks-for-action","repo_url":"https://github.com/feichtenhofer/twostreamfusion","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"none","reach":{"status":"ok","spdx":"NOASSERTION"}},{"paper_slug":"two-stream-convolutional-networks-for-action","repo_url":"https://github.com/HsinYingLee/OPN","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"caffe2","reach":null},{"paper_slug":"two-stream-convolutional-networks-for-action","repo_url":"https://github.com/Michaelgod/test","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":null},{"paper_slug":"two-stream-convolutional-networks-for-action","repo_url":"https://github.com/damien911224/theWorldInSafety","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok","spdx":"GPL-3.0"}},{"paper_slug":"two-stream-convolutional-networks-for-action","repo_url":"https://github.com/jerryljq/ActionRecognition","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":null},{"paper_slug":"two-stream-convolutional-networks-for-action","repo_url":"https://github.com/mcgridles/LENS","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"two-stream-convolutional-networks-for-action","repo_url":"https://github.com/woodfrog/ActionRecognition","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"action-classification","task_name":"Action Classification"},{"task_slug":"action-recognition-in-videos","task_name":"Action Recognition"},{"task_slug":"action-recognition-in-videos-2","task_name":"Action Recognition In Videos"},{"task_slug":"classification","task_name":"General Classification"},{"task_slug":"multi-task-learning","task_name":"Multi-Task Learning"},{"task_slug":"optical-flow-estimation","task_name":"Optical Flow Estimation"},{"task_slug":"action-recognition","task_name":"Temporal Action Localization"},{"task_slug":"video-classification","task_name":"Video Classification"},{"task_slug":"two","task_name":"Vocal Bursts Valence Prediction"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/action-classification-on-charades","task":"Action Classification","dataset":"Charades","model":"2-Strm","rank_in_archive_order":48,"of":49,"metrics":{"MAP":"18.6"},"uses_additional_data":false},{"leaderboard":"/sota/action-recognition-in-videos-on-hmdb-51","task":"Action Recognition","dataset":"HMDB-51","model":"Two-Stream (ImageNet pretrained)","rank_in_archive_order":68,"of":77,"metrics":{"Average accuracy of 3 splits":"59.4"},"uses_additional_data":true},{"leaderboard":"/sota/action-recognition-in-videos-on-ucf101","task":"Action Recognition","dataset":"UCF101","model":"Two-Stream (ImageNet pretrained)","rank_in_archive_order":75,"of":91,"metrics":{"3-fold Accuracy":"88.0"},"uses_additional_data":true},{"leaderboard":"/sota/hand-gesture-recognition-on-viva-hand-1","task":"Hand Gesture Recognition","dataset":"VIVA Hand Gestures Dataset","model":"Two Stream CNNs","rank_in_archive_order":3,"of":3,"metrics":{"Accuracy":"68"},"uses_additional_data":false}],"syntology":{"syntology_url":null,"atlas_url":"https://app.syntology.ai/?focus=1406.2199","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1406.2199"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/feichtenhofer/twostreamfusion","reach":{"status":"ok","spdx":"NOASSERTION"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/HsinYingLee/OPN","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/jerryljq/ActionRecognition","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/mcgridles/LENS","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Michaelgod/test","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/woodfrog/ActionRecognition","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/damien911224/theWorldInSafety","reach":{"status":"ok","spdx":"GPL-3.0"}}],"summary":{"ran_draft_wrong":1,"unverified":6},"by_repo_kind":{"listed":{"samples":7,"ran":1,"repositories":3}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":2,"samples":[{"code_sha256_prefix":"fc2bbda446121dde","entry":"decode_prediction","repo":"Michaelgod/test","repo_kind":"listed","path":"predict.py","file_url":"https://github.com/Michaelgod/test/blob/HEAD/predict.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"mcp_get_code":{"code_sha256":"fc2bbda446121dde"}},{"code_sha256_prefix":"5e7a557c615eb17e","entry":"conv_block","repo":"woodfrog/ActionRecognition","repo_kind":"listed","path":"models/resnet50.py","file_url":"https://github.com/woodfrog/ActionRecognition/blob/HEAD/models/resnet50.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"5e7a557c615eb17e"}},{"code_sha256_prefix":"c064c9e4cc5ca645","entry":"decode_prediction","repo":"jerryljq/ActionRecognition","repo_kind":"listed","path":"predict.py","file_url":"https://github.com/jerryljq/ActionRecognition/blob/HEAD/predict.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"c064c9e4cc5ca645"}},{"code_sha256_prefix":"4679231cd26e9089","entry":"identity_block","repo":"woodfrog/ActionRecognition","repo_kind":"listed","path":"models/resnet50.py","file_url":"https://github.com/woodfrog/ActionRecognition/blob/HEAD/models/resnet50.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"4679231cd26e9089"}},{"code_sha256_prefix":"b1aaeed051544edb","entry":"preprocess_input","repo":"woodfrog/ActionRecognition","repo_kind":"listed","path":"rnn_practice/LRCN/imagenet_utils.py","file_url":"https://github.com/woodfrog/ActionRecognition/blob/HEAD/rnn_practice/LRCN/imagenet_utils.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"b1aaeed051544edb"}},{"code_sha256_prefix":"4953cb1be1c7abc8","entry":"preprocess_single_frame","repo":"woodfrog/ActionRecognition","repo_kind":"listed","path":"predict.py","file_url":"https://github.com/woodfrog/ActionRecognition/blob/HEAD/predict.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"4953cb1be1c7abc8"}},{"code_sha256_prefix":"0e464771c030663d","entry":"temporal_CNN","repo":"woodfrog/ActionRecognition","repo_kind":"listed","path":"models/temporal_CNN.py","file_url":"https://github.com/woodfrog/ActionRecognition/blob/HEAD/models/temporal_CNN.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"0e464771c030663d"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}