{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/action-recognition-in-videos/papers/10","list_of":"/task/action-recognition-in-videos","task":"Action Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":10,"pages_in_order":28,"rows_per_page":100,"rows":[901,1000],"of":2759,"counts":{"archive_papers_tagged":2759,"with_a_code_link":1058,"where_syntology_ran_a_sample":275,"not_listed_spam_title":0,"listed":2759,"listed_where_code_ran":275,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":232,"every_run_a_failure_of_syntologys_instrument":43,"listed_with_a_run_with_no_instrument_failure":232,"listed_every_run_a_failure_of_syntologys_instrument":43,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/action-recognition-in-videos","prev":"/task/action-recognition-in-videos/papers/9","next":"/task/action-recognition-in-videos/papers/11","papers":[{"url":"/paper/listen-to-look-action-recognition-by","slug":"listen-to-look-action-recognition-by","title":"Listen to Look: Action Recognition by Previewing Audio","date":"2019-12-10","arxiv_id":"1912.04487","repositories_listed":1,"syntology":null},{"url":"/paper/synthetic-humans-for-action-recognition-from","slug":"synthetic-humans-for-action-recognition-from","title":"Synthetic Humans for Action Recognition from Unseen Viewpoints","date":"2019-12-09","arxiv_id":"1912.04070","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-pyramid-network-for-video-domain","slug":"adversarial-pyramid-network-for-video-domain","title":"VideoDG: Generalizing Temporal Relations in Videos to Novel Domains","date":"2019-12-08","arxiv_id":"1912.03716","repositories_listed":1,"syntology":null},{"url":"/paper/more-is-less-learning-efficient-video-1","slug":"more-is-less-learning-efficient-video-1","title":"More Is Less: Learning Efficient Video Representations by Big-Little Network and Depthwise Temporal Aggregation","date":"2019-12-02","arxiv_id":"1912.00869","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/more-is-less-learning-efficient-video-1#ran","syntology_url":"https://syntology.ai/paper/1912.00869","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.00869"}},"official":{"repos":["IBM/bLVNet-TAM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/self-supervised-learning-by-cross-modal-audio","slug":"self-supervised-learning-by-cross-modal-audio","title":"Self-Supervised Learning by Cross-Modal Audio-Video Clustering","date":"2019-11-28","arxiv_id":"1911.12667","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/self-supervised-learning-by-cross-modal-audio#ran","syntology_url":"https://syntology.ai/paper/1911.12667","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.12667"}},"official":{"repos":["HumamAlwassel/XDC"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/predict-cluster-unsupervised-skeleton-based","slug":"predict-cluster-unsupervised-skeleton-based","title":"PREDICT & CLUSTER: Unsupervised Skeleton Based Action Recognition","date":"2019-11-27","arxiv_id":"1911.12409","repositories_listed":1,"syntology":null},{"url":"/paper/mmtm-multimodal-transfer-module-for-cnn","slug":"mmtm-multimodal-transfer-module-for-cnn","title":"MMTM: Multimodal Transfer Module for CNN Fusion","date":"2019-11-20","arxiv_id":"1911.08670","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mmtm-multimodal-transfer-module-for-cnn#ran","syntology_url":"https://syntology.ai/paper/1911.08670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.08670"}},"official":null}},{"url":"/paper/action-recognition-using-volumetric-motion","slug":"action-recognition-using-volumetric-motion","title":"Action Recognition Using Volumetric Motion Representations","date":"2019-11-19","arxiv_id":"1911.08511","repositories_listed":1,"syntology":null},{"url":"/paper/multi-attention-networks-for-temporal","slug":"multi-attention-networks-for-temporal","title":"Multi-attention Networks for Temporal Localization of Video-level Labels","date":"2019-11-15","arxiv_id":"1911.06866","repositories_listed":1,"syntology":null},{"url":"/paper/rwf-2000-an-open-large-scale-video-database","slug":"rwf-2000-an-open-large-scale-video-database","title":"RWF-2000: An Open Large Scale Video Database for Violence Detection","date":"2019-11-14","arxiv_id":"1911.05913","repositories_listed":1,"syntology":null},{"url":"/paper/guided-weak-supervision-for-action","slug":"guided-weak-supervision-for-action","title":"Guided Weak Supervision for Action Recognition with Scarce Data to Assess Skills of Children with Autism","date":"2019-11-11","arxiv_id":"1911.04140","repositories_listed":1,"syntology":null},{"url":"/paper/learning-graph-convolutional-network-for","slug":"learning-graph-convolutional-network-for","title":"Learning Graph Convolutional Network for Skeleton-based Human Action Recognition by Neural Searching","date":"2019-11-11","arxiv_id":"1911.04131","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-graph-convolutional-network-for#ran","syntology_url":"https://syntology.ai/paper/1911.04131","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04131"}},"official":null}},{"url":"/paper/human-action-recognition-using-deep","slug":"human-action-recognition-using-deep","title":"Human Action Recognition Using Deep Multilevel Multimodal (M2) Fusion of Depth and Inertial Sensors","date":"2019-10-25","arxiv_id":"1910.11482","repositories_listed":1,"syntology":null},{"url":"/paper/spatiotemporal-tile-based-attention-guided","slug":"spatiotemporal-tile-based-attention-guided","title":"Spatiotemporal Tile-based Attention-guided LSTMs for Traffic Video Prediction","date":"2019-10-24","arxiv_id":"1910.11030","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/spatiotemporal-tile-based-attention-guided#ran","syntology_url":"https://syntology.ai/paper/1910.11030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.11030"}},"official":{"repos":["tumeteor/neurips2019challenge"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/conquering-the-cnn-over-parameterization","slug":"conquering-the-cnn-over-parameterization","title":"Volterra Neural Networks (VNNs)","date":"2019-10-21","arxiv_id":"1910.09616","repositories_listed":1,"syntology":null},{"url":"/paper/making-third-person-techniques-recognize","slug":"making-third-person-techniques-recognize","title":"Making Third Person Techniques Recognize First-Person Actions in Egocentric Videos","date":"2019-10-17","arxiv_id":"1910.07766","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-and-iteratively-improving-recurrent","slug":"adaptive-and-iteratively-improving-recurrent","title":"Adaptive and Iteratively Improving Recurrent Lateral Connections","date":"2019-10-16","arxiv_id":"1910.11105","repositories_listed":1,"syntology":null},{"url":"/paper/seeing-and-hearing-egocentric-actions-how","slug":"seeing-and-hearing-egocentric-actions-how","title":"Seeing and Hearing Egocentric Actions: How Much Can We Learn?","date":"2019-10-15","arxiv_id":"1910.06693","repositories_listed":1,"syntology":null},{"url":"/paper/context-gated-convolution","slug":"context-gated-convolution","title":"Context-Gated Convolution","date":"2019-10-12","arxiv_id":"1910.05577","repositories_listed":1,"syntology":null},{"url":"/paper/interaction-relational-network-for-mutual","slug":"interaction-relational-network-for-mutual","title":"Interaction Relational Network for Mutual Action Recognition","date":"2019-10-11","arxiv_id":"1910.04963","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/interaction-relational-network-for-mutual#ran","syntology_url":"https://syntology.ai/paper/1910.04963","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.04963"}},"official":{"repos":["mauriciolp/inter-rel-net"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/graph-based-spatial-temporal-feature-learning","slug":"graph-based-spatial-temporal-feature-learning","title":"Graph-based Spatial-temporal Feature Learning for Neuromorphic Vision Sensing","date":"2019-10-08","arxiv_id":"1910.03579","repositories_listed":1,"syntology":null},{"url":"/paper/grouped-spatial-temporal-aggregation-for","slug":"grouped-spatial-temporal-aggregation-for","title":"Grouped Spatial-Temporal Aggregation for Efficient Action Recognition","date":"2019-09-28","arxiv_id":"1909.13130","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/grouped-spatial-temporal-aggregation-for#ran","syntology_url":"https://syntology.ai/paper/1909.13130","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.13130"}},"official":null}},{"url":"/paper/learning-deep-representations-for-video-based","slug":"learning-deep-representations-for-video-based","title":"Learning deep representations for video-based intake gesture detection","date":"2019-09-24","arxiv_id":"1909.10695","repositories_listed":1,"syntology":null},{"url":"/paper/class-feature-pyramids-for-video-explanation","slug":"class-feature-pyramids-for-video-explanation","title":"Class Feature Pyramids for Video Explanation","date":"2019-09-18","arxiv_id":"1909.08611","repositories_listed":1,"syntology":null},{"url":"/paper/multiple-human-tracking-using-multi-cues","slug":"multiple-human-tracking-using-multi-cues","title":"Multiple Human Tracking using Multi-Cues including Primitive Action Features","date":"2019-09-18","arxiv_id":"1909.08171","repositories_listed":1,"syntology":null},{"url":"/paper/deep-point-wise-prediction-for-action","slug":"deep-point-wise-prediction-for-action","title":"Deep Point-wise Prediction for Action Temporal Proposal","date":"2019-09-17","arxiv_id":"1909.07725","repositories_listed":1,"syntology":null},{"url":"/paper/comparative-analysis-of-cnn-based","slug":"comparative-analysis-of-cnn-based","title":"Comparative Analysis of CNN-based Spatiotemporal Reasoning in Videos","date":"2019-09-11","arxiv_id":"1909.05165","repositories_listed":1,"syntology":null},{"url":"/paper/skeleton-image-representation-for-3d-action","slug":"skeleton-image-representation-for-3d-action","title":"Skeleton Image Representation for 3D Action Recognition based on Tree Structure and Reference Joints","date":"2019-09-11","arxiv_id":"1909.05704","repositories_listed":1,"syntology":null},{"url":"/paper/video-representation-learning-by-dense","slug":"video-representation-learning-by-dense","title":"Video Representation Learning by Dense Predictive Coding","date":"2019-09-10","arxiv_id":"1909.04656","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/video-representation-learning-by-dense#ran","syntology_url":"https://syntology.ai/paper/1909.04656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.04656"}},"official":{"repos":["TengdaHan/DPC"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/tensor-analysis-with-n-mode-generalized","slug":"tensor-analysis-with-n-mode-generalized","title":"Tensor Analysis with n-Mode Generalized Difference Subspace","date":"2019-09-04","arxiv_id":"1909.01954","repositories_listed":1,"syntology":null},{"url":"/paper/epic-fusion-audio-visual-temporal-binding-for","slug":"epic-fusion-audio-visual-temporal-binding-for","title":"EPIC-Fusion: Audio-Visual Temporal Binding for Egocentric Action Recognition","date":"2019-08-22","arxiv_id":"1908.08498","repositories_listed":1,"syntology":null},{"url":"/paper/i3d-lstm-a-new-model-for-human-action","slug":"i3d-lstm-a-new-model-for-human-action","title":"I3D-LSTM: A New Model for Human Action Recognition","date":"2019-08-09","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/open-set-domain-adaptation-for-image-and","slug":"open-set-domain-adaptation-for-image-and","title":"Open Set Domain Adaptation for Image and Action Recognition","date":"2019-07-30","arxiv_id":"1907.12865","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/open-set-domain-adaptation-for-image-and#ran","syntology_url":"https://syntology.ai/paper/1907.12865","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.12865"}},"official":{"repos":["Heliot7/open-set-da"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/skelemotion-a-new-representation-of-skeleton","slug":"skelemotion-a-new-representation-of-skeleton","title":"SkeleMotion: A New Representation of Skeleton Joint Sequences Based on Motion Information for 3D Action Recognition","date":"2019-07-30","arxiv_id":"1907.13025","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/skelemotion-a-new-representation-of-skeleton#ran","syntology_url":"https://syntology.ai/paper/1907.13025","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.13025"}},"official":{"repos":["carloscaetano/skeleton-images"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-learning-for-optical-flow","slug":"unsupervised-learning-for-optical-flow","title":"Unsupervised Learning for Optical Flow Estimation Using Pyramid Convolution LSTM","date":"2019-07-26","arxiv_id":"1907.11628","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unsupervised-learning-for-optical-flow#ran","syntology_url":"https://syntology.ai/paper/1907.11628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.11628"}},"official":{"repos":["Kwanss/PCLNet"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-visual-actions-using-multiple-verb","slug":"learning-visual-actions-using-multiple-verb","title":"Learning Visual Actions Using Multiple Verb-Only Labels","date":"2019-07-25","arxiv_id":"1907.11117","repositories_listed":1,"syntology":null},{"url":"/paper/i-know-the-relationships-zero-shot-action","slug":"i-know-the-relationships-zero-shot-action","title":"I Know the Relationships: Zero-Shot Action Recognition via Two-Stream Graph Convolutional Networks and Knowledge Graphs","date":"2019-07-17","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/non-local-graph-convolutional-networks-for-1","slug":"non-local-graph-convolutional-networks-for-1","title":"Non-Local Graph Convolutional Networks for Skeleton-Based Action Recognition","date":"2019-07-04","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-comparative-review-of-recent-kinect-based","slug":"a-comparative-review-of-recent-kinect-based","title":"A Comparative Review of Recent Kinect-based Action Recognition Algorithms","date":"2019-06-24","arxiv_id":"1906.09955","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-learning-of-object-structure-and","slug":"unsupervised-learning-of-object-structure-and","title":"Unsupervised Learning of Object Structure and Dynamics from Videos","date":"2019-06-19","arxiv_id":"1906.07889","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-transformer-networks-joint-learning-1","slug":"temporal-transformer-networks-joint-learning-1","title":"Temporal Transformer Networks: Joint Learning of Invariant and Discriminative Time Warping","date":"2019-06-13","arxiv_id":"1906.05947","repositories_listed":1,"syntology":null},{"url":"/paper/recognizing-manipulation-actions-from-state","slug":"recognizing-manipulation-actions-from-state","title":"Recognizing Manipulation Actions from State-Transformations","date":"2019-06-12","arxiv_id":"1906.05147","repositories_listed":1,"syntology":null},{"url":"/paper/detecting-the-starting-frame-of-actions-in","slug":"detecting-the-starting-frame-of-actions-in","title":"Detecting the Starting Frame of Actions in Video","date":"2019-06-07","arxiv_id":"1906.03340","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-autoregressive-video-models","slug":"scaling-autoregressive-video-models","title":"Scaling Autoregressive Video Models","date":"2019-06-06","arxiv_id":"1906.02634","repositories_listed":1,"syntology":null},{"url":"/paper/mining-youtube-a-dataset-for-learning-fine","slug":"mining-youtube-a-dataset-for-learning-fine","title":"Mining YouTube - A dataset for learning fine-grained action concepts from webly supervised video data","date":"2019-06-03","arxiv_id":"1906.01012","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-hierarchical-dynamic-model-for-human","slug":"bayesian-hierarchical-dynamic-model-for-human","title":"Bayesian Hierarchical Dynamic Model for Human Action Recognition","date":"2019-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/collaborative-spatiotemporal-feature-learning","slug":"collaborative-spatiotemporal-feature-learning","title":"Collaborative Spatiotemporal Feature Learning for Video Action Recognition","date":"2019-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mars-motion-augmented-rgb-stream-for-action","slug":"mars-motion-augmented-rgb-stream-for-action","title":"MARS: Motion-Augmented RGB Stream for Action Recognition","date":"2019-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mfas-multimodal-fusion-architecture-search-1","slug":"mfas-multimodal-fusion-architecture-search-1","title":"MFAS: Multimodal Fusion Architecture Search","date":"2019-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/skeleton-based-action-recognition-with-4","slug":"skeleton-based-action-recognition-with-4","title":"Skeleton-Based Action Recognition With Directed Graph Neural Networks","date":"2019-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-learning-from-video-with-deep","slug":"unsupervised-learning-from-video-with-deep","title":"Unsupervised Learning from Video with Deep Neural Embeddings","date":"2019-05-28","arxiv_id":"1905.11954","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unsupervised-learning-from-video-with-deep#ran","syntology_url":"https://syntology.ai/paper/1905.11954","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.11954"}},"official":{"repos":["neuroailab/VIE"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-temporal-information-for-improved","slug":"exploring-temporal-information-for-improved","title":"Exploring Temporal Information for Improved Video Understanding","date":"2019-05-25","arxiv_id":"1905.10654","repositories_listed":1,"syntology":null},{"url":"/paper/lightweight-network-architecture-for-real","slug":"lightweight-network-architecture-for-real","title":"Lightweight Network Architecture for Real-Time Action Recognition","date":"2019-05-21","arxiv_id":"1905.08711","repositories_listed":1,"syntology":null},{"url":"/paper/actional-structural-graph-convolutional","slug":"actional-structural-graph-convolutional","title":"Actional-Structural Graph Convolutional Networks for Skeleton-based Action Recognition","date":"2019-04-26","arxiv_id":"1904.12659","repositories_listed":1,"syntology":{"n":18,"n_ran":14,"n_constructed":0,"n_ran_checked":12,"n_instrument":2,"n_unverified":4,"n_honours":2,"n_violates":1,"n_no_contract":9,"n_pointer_only":5,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 2 honoured, 1 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/actional-structural-graph-convolutional#ran","syntology_url":"https://syntology.ai/paper/1904.12659","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.12659"}},"official":{"repos":["limaosen0/AS-GCN"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/holistic-large-scale-video-understanding","slug":"holistic-large-scale-video-understanding","title":"Large Scale Holistic Video Understanding","date":"2019-04-25","arxiv_id":"1904.11451","repositories_listed":1,"syntology":null},{"url":"/paper/190412602","slug":"190412602","title":"EV-Action: Electromyography-Vision Multi-Modal Action Dataset","date":"2019-04-20","arxiv_id":"1904.12602","repositories_listed":1,"syntology":null},{"url":"/paper/190411953","slug":"190411953","title":"Temporal Unet: Sample Level Human Action Recognition using WiFi","date":"2019-04-19","arxiv_id":"1904.11953","repositories_listed":1,"syntology":null},{"url":"/paper/simple-yet-efficient-real-time-pose-based","slug":"simple-yet-efficient-real-time-pose-based","title":"Simple yet efficient real-time pose-based action recognition","date":"2019-04-19","arxiv_id":"1904.09140","repositories_listed":1,"syntology":null},{"url":"/paper/step-spatio-temporal-progressive-learning-for","slug":"step-spatio-temporal-progressive-learning-for","title":"STEP: Spatio-Temporal Progressive Learning for Video Action Detection","date":"2019-04-19","arxiv_id":"1904.09288","repositories_listed":1,"syntology":null},{"url":"/paper/out-of-distribution-detection-for-generalized","slug":"out-of-distribution-detection-for-generalized","title":"Out-of-Distribution Detection for Generalized Zero-Shot Action Recognition","date":"2019-04-18","arxiv_id":"1904.08703","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/out-of-distribution-detection-for-generalized#ran","syntology_url":"https://syntology.ai/paper/1904.08703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.08703"}},"official":{"repos":["naraysa/gzsl-od"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/190407911","slug":"190407911","title":"REPAIR: Removing Representation Bias by Dataset Resampling","date":"2019-04-16","arxiv_id":"1904.07911","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/190407911#ran","syntology_url":"https://syntology.ai/paper/1904.07911","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.07911"}},"official":{"repos":["JerryYLi/Dataset-REPAIR"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/audio-visual-model-distillation-using","slug":"audio-visual-model-distillation-using","title":"Audio-Visual Model Distillation Using Acoustic Images","date":"2019-04-16","arxiv_id":"1904.07933","repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-space-time-graphs-for-video","slug":"recurrent-space-time-graphs-for-video","title":"Recurrent Space-time Graph Neural Networks","date":"2019-04-11","arxiv_id":"1904.05582","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recurrent-space-time-graphs-for-video#ran","syntology_url":"https://syntology.ai/paper/1904.05582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.05582"}},"official":{"repos":["IuliaDuta/RSTG"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/action-recognition-from-single-timestamp","slug":"action-recognition-from-single-timestamp","title":"Action Recognition from Single Timestamp Supervision in Untrimmed Videos","date":"2019-04-09","arxiv_id":"1904.04689","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-spatio-temporal","slug":"self-supervised-spatio-temporal","title":"Self-supervised Spatio-temporal Representation Learning for Videos by Predicting Motion and Appearance Statistics","date":"2019-04-07","arxiv_id":"1904.03597","repositories_listed":1,"syntology":null},{"url":"/paper/two-stream-oriented-video-super-resolution","slug":"two-stream-oriented-video-super-resolution","title":"Two-Stream Action Recognition-Oriented Video Super-Resolution","date":"2019-03-13","arxiv_id":"1903.05577","repositories_listed":1,"syntology":null},{"url":"/paper/collaborative-spatio-temporal-feature","slug":"collaborative-spatio-temporal-feature","title":"Collaborative Spatio-temporal Feature Learning for Video Action Recognition","date":"2019-03-04","arxiv_id":"1903.01197","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-the-self-using-grounded-affordances-to","slug":"beyond-the-self-using-grounded-affordances-to","title":"Beyond the Self: Using Grounded Affordances to Interpret and Describe Others' Actions","date":"2019-02-26","arxiv_id":"1902.09705","repositories_listed":1,"syntology":null},{"url":"/paper/uav-gesture-a-dataset-for-uav-control-and","slug":"uav-gesture-a-dataset-for-uav-control-and","title":"UAV-GESTURE: A Dataset for UAV Control and Gesture Recognition","date":"2019-01-09","arxiv_id":"1901.02602","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-learning-by-hallucinating-missing","slug":"cross-modal-learning-by-hallucinating-missing","title":"Cross-modal Learning by Hallucinating Missing Modalities in RGB-D Vision","date":"2019-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/d3d-distilled-3d-networks-for-video-action","slug":"d3d-distilled-3d-networks-for-video-action","title":"D3D: Distilled 3D Networks for Video Action Recognition","date":"2018-12-19","arxiv_id":"1812.08249","repositories_listed":1,"syntology":null},{"url":"/paper/improving-the-performance-of-unimodal-dynamic","slug":"improving-the-performance-of-unimodal-dynamic","title":"Improving the Performance of Unimodal Dynamic Hand-Gesture Recognition with Multimodal Training","date":"2018-12-14","arxiv_id":"1812.06145","repositories_listed":1,"syntology":null},{"url":"/paper/structure-aware-convolutional-neural-networks","slug":"structure-aware-convolutional-neural-networks","title":"Structure-Aware Convolutional Neural Networks","date":"2018-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/lsta-long-short-term-attention-for-egocentric","slug":"lsta-long-short-term-attention-for-egocentric","title":"LSTA: Long Short-Term Attention for Egocentric Action Recognition","date":"2018-11-26","arxiv_id":"1811.10698","repositories_listed":1,"syntology":null},{"url":"/paper/compressing-recurrent-neural-networks-with","slug":"compressing-recurrent-neural-networks-with","title":"Compressing Recurrent Neural Networks with Tensor Ring for Action Recognition","date":"2018-11-19","arxiv_id":"1811.07503","repositories_listed":1,"syntology":null},{"url":"/paper/a-perceptual-prediction-framework-for-self","slug":"a-perceptual-prediction-framework-for-self","title":"A Perceptual Prediction Framework for Self Supervised Event Segmentation","date":"2018-11-12","arxiv_id":"1811.04869","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/a-perceptual-prediction-framework-for-self#ran","syntology_url":"https://syntology.ai/paper/1811.04869","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.04869"}},"official":{"repos":["CVPRUSFTampa/EventSegmentation"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/cross-and-learn-cross-modal-self-supervision","slug":"cross-and-learn-cross-modal-self-supervision","title":"Cross and Learn: Cross-Modal Self-Supervision","date":"2018-11-09","arxiv_id":"1811.03879","repositories_listed":1,"syntology":null},{"url":"/paper/deepgru-deep-gesture-recognition-utility","slug":"deepgru-deep-gesture-recognition-utility","title":"DeepGRU: Deep Gesture Recognition Utility","date":"2018-10-30","arxiv_id":"1810.12514","repositories_listed":1,"syntology":null},{"url":"/paper/learning-with-privileged-information-via","slug":"learning-with-privileged-information-via","title":"Learning with privileged information via adversarial discriminative modality distillation","date":"2018-10-19","arxiv_id":"1810.08437","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/learning-with-privileged-information-via#ran","syntology_url":"https://syntology.ai/paper/1810.08437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.08437"}},"official":{"repos":["pmorerio/admd"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/cross-modal-and-hierarchical-modeling-of","slug":"cross-modal-and-hierarchical-modeling-of","title":"Cross-Modal and Hierarchical Modeling of Video and Text","date":"2018-10-16","arxiv_id":"1810.07212","repositories_listed":1,"syntology":null},{"url":"/paper/towards-high-resolution-video-generation-with","slug":"towards-high-resolution-video-generation-with","title":"Towards High Resolution Video Generation with Progressive Growing of Sliced Wasserstein GANs","date":"2018-10-04","arxiv_id":"1810.02419","repositories_listed":1,"syntology":null},{"url":"/paper/rate-accuracy-trade-off-in-video","slug":"rate-accuracy-trade-off-in-video","title":"Rate-Accuracy Trade-Off In Video Classification With Deep Convolutional Neural Networks","date":"2018-09-27","arxiv_id":"1810.03964","repositories_listed":1,"syntology":null},{"url":"/paper/part-based-graph-convolutional-network-for","slug":"part-based-graph-convolutional-network-for","title":"Part-based Graph Convolutional Network for Action Recognition","date":"2018-09-13","arxiv_id":"1809.04983","repositories_listed":1,"syntology":null},{"url":"/paper/using-phase-instead-of-optical-flow-for","slug":"using-phase-instead-of-optical-flow-for","title":"Using phase instead of optical flow for action recognition","date":"2018-09-10","arxiv_id":"1809.03258","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-learning-of-view-invariant","slug":"unsupervised-learning-of-view-invariant","title":"Unsupervised Learning of View-invariant Action Representations","date":"2018-09-06","arxiv_id":"1809.01844","repositories_listed":1,"syntology":null},{"url":"/paper/arbee-towards-automated-recognition-of-bodily","slug":"arbee-towards-automated-recognition-of-bodily","title":"ARBEE: Towards Automated Recognition of Bodily Expression of Emotion In the Wild","date":"2018-08-28","arxiv_id":"1808.09568","repositories_listed":1,"syntology":null},{"url":"/paper/targeted-nonlinear-adversarial-perturbations","slug":"targeted-nonlinear-adversarial-perturbations","title":"Targeted Nonlinear Adversarial Perturbations in Images and Videos","date":"2018-08-27","arxiv_id":"1809.00958","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-human-human-interactions-a","slug":"understanding-human-human-interactions-a","title":"Analyzing Human-Human Interactions: A Survey","date":"2018-07-31","arxiv_id":"1808.00022","repositories_listed":1,"syntology":null},{"url":"/paper/actor-centric-relation-network","slug":"actor-centric-relation-network","title":"Actor-Centric Relation Network","date":"2018-07-28","arxiv_id":"1807.10982","repositories_listed":1,"syntology":null},{"url":"/paper/a-variational-time-series-feature-extractor","slug":"a-variational-time-series-feature-extractor","title":"A Variational Time Series Feature Extractor for Action Prediction","date":"2018-07-06","arxiv_id":"1807.02350","repositories_listed":1,"syntology":null},{"url":"/paper/action-recognition-for-depth-video-using","slug":"action-recognition-for-depth-video-using","title":"Action Recognition for Depth Video using Multi-view Dynamic Images","date":"2018-06-29","arxiv_id":"1806.11269","repositories_listed":1,"syntology":null},{"url":"/paper/modality-distillation-with-multiple-stream","slug":"modality-distillation-with-multiple-stream","title":"Modality Distillation with Multiple Stream Networks for Action Recognition","date":"2018-06-19","arxiv_id":"1806.07110","repositories_listed":1,"syntology":null},{"url":"/paper/actor-and-observer-joint-modeling-of-first","slug":"actor-and-observer-joint-modeling-of-first","title":"Actor and Observer: Joint Modeling of First and Third-Person Videos","date":"2018-04-25","arxiv_id":"1804.09627","repositories_listed":1,"syntology":null},{"url":"/paper/memory-attention-networks-for-skeleton-based","slug":"memory-attention-networks-for-skeleton-based","title":"Memory Attention Networks for Skeleton-based Action Recognition","date":"2018-04-23","arxiv_id":"1804.08254","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/memory-attention-networks-for-skeleton-based#ran","syntology_url":"https://syntology.ai/paper/1804.08254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.08254"}},"official":null}},{"url":"/paper/stair-actions-a-video-dataset-of-everyday","slug":"stair-actions-a-video-dataset-of-everyday","title":"STAIR Actions: A Video Dataset of Everyday Home Actions","date":"2018-04-12","arxiv_id":"1804.04326","repositories_listed":1,"syntology":null},{"url":"/paper/audio-visual-scene-analysis-with-self","slug":"audio-visual-scene-analysis-with-self","title":"Audio-Visual Scene Analysis with Self-Supervised Multisensory Features","date":"2018-04-10","arxiv_id":"1804.03641","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-learning-of-motion-representation","slug":"end-to-end-learning-of-motion-representation","title":"End-to-End Learning of Motion Representation for Video Understanding","date":"2018-04-02","arxiv_id":"1804.00413","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/end-to-end-learning-of-motion-representation#ran","syntology_url":"https://syntology.ai/paper/1804.00413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.00413"}},"official":null}},{"url":"/paper/deja-vu-motion-prediction-in-static-images","slug":"deja-vu-motion-prediction-in-static-images","title":"Deja Vu: Motion Prediction in Static Images","date":"2018-03-19","arxiv_id":"1803.06951","repositories_listed":1,"syntology":null},{"url":"/paper/analysis-of-hand-segmentation-in-the-wild","slug":"analysis-of-hand-segmentation-in-the-wild","title":"Analysis of Hand Segmentation in the Wild","date":"2018-03-08","arxiv_id":"1803.03317","repositories_listed":1,"syntology":null},{"url":"/paper/glimpse-clouds-human-activity-recognition","slug":"glimpse-clouds-human-activity-recognition","title":"Glimpse Clouds: Human Activity Recognition from Unstructured Feature Points","date":"2018-02-22","arxiv_id":"1802.07898","repositories_listed":1,"syntology":null}],"record_sha256":"66dc3a1302630eb857350e98fb23f25c29979acf34ed5a27a65dedf87ea353d5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}