{"url":"/task/action-recognition","name":"Temporal Action Localization","slug":"action-recognition","description_markdown":"Temporal Action Localization aims to detect activities in the video stream and  output beginning and end timestamps. It is closely related to  Temporal Action Proposal Generation.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":1477,"papers_with_code":493,"benchmarks":14,"benchmark_tables_in_archive":14,"benchmark_tables_shown":14,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":42,"subtasks":8,"parent_tasks":3},"benchmarks":[{"leaderboard":"/sota/temporal-action-localization-on-thumos14","slug":"temporal-action-localization-on-thumos14","dataset":"THUMOS’14","dataset_url":"/dataset/thumos14-1","rows_in_archive":42,"metrics":["Avg mAP (0.3:0.7)","mAP IOU@0.1","mAP IOU@0.2","mAP IOU@0.3","mAP IOU@0.4","mAP IOU@0.5","mAP IOU@0.6","mAP IOU@0.7"],"first_row_in_archive_order":{"model":"AdaTAD (VideoMAEv2-giant)","paper_title":"End-to-End Temporal Action Detection with 1B Parameters Across 1000 Frames","paper_url":"/paper/end-to-end-temporal-action-detection-with-1b","paper_date":"2023-11-28","arxiv_id":"2311.17241","code_links":[{"title":"sming256/OpenTAD","url":"https://github.com/sming256/OpenTAD"},{"title":"sming256/AdaTAD","url":"https://github.com/sming256/AdaTAD"}],"syntology":null}},{"leaderboard":"/sota/temporal-action-localization-on-activitynet","slug":"temporal-action-localization-on-activitynet","dataset":"ActivityNet-1.3","dataset_url":"/dataset/activitynet","rows_in_archive":33,"metrics":["mAP","mAP IOU@0.5","mAP IOU@0.75","mAP IOU@0.95"],"first_row_in_archive_order":{"model":"RDFA-S6 (InternVideo2-6B)","paper_title":"Enhancing Temporal Action Localization: Advanced S6 Modeling with Recurrent Mechanism","paper_url":"/paper/enhancing-temporal-action-localization","paper_date":"2024-07-18","arxiv_id":"2407.13078","code_links":[{"title":"lsy0882/RDFA-S6","url":"https://github.com/lsy0882/RDFA-S6"}],"syntology":null}},{"leaderboard":"/sota/temporal-action-localization-on-hacs","slug":"temporal-action-localization-on-hacs","dataset":"HACS","dataset_url":"/dataset/hacs","rows_in_archive":12,"metrics":["Average-mAP","mAP@0.5","mAP@0.75","mAP@0.95"],"first_row_in_archive_order":{"model":"RDFA-S6 (InternVideo2-6B)","paper_title":"Enhancing Temporal Action Localization: Advanced S6 Modeling with Recurrent Mechanism","paper_url":"/paper/enhancing-temporal-action-localization","paper_date":"2024-07-18","arxiv_id":"2407.13078","code_links":[{"title":"lsy0882/RDFA-S6","url":"https://github.com/lsy0882/RDFA-S6"}],"syntology":null}},{"leaderboard":"/sota/temporal-action-localization-on-fineaction","slug":"temporal-action-localization-on-fineaction","dataset":"FineAction","dataset_url":"/dataset/fineaction","rows_in_archive":9,"metrics":["mAP","mAP IOU@0.5","mAP IOU@0.75","mAP IOU@0.95"],"first_row_in_archive_order":{"model":"RDFA-S6 (InternVideo2-6B)","paper_title":"Enhancing Temporal Action Localization: Advanced S6 Modeling with Recurrent Mechanism","paper_url":"/paper/enhancing-temporal-action-localization","paper_date":"2024-07-18","arxiv_id":"2407.13078","code_links":[{"title":"lsy0882/RDFA-S6","url":"https://github.com/lsy0882/RDFA-S6"}],"syntology":null}},{"leaderboard":"/sota/temporal-action-localization-on-multithumos-1","slug":"temporal-action-localization-on-multithumos-1","dataset":"MultiTHUMOS","dataset_url":"/dataset/multithumos","rows_in_archive":8,"metrics":["Average mAP","mAP IOU@0.1","mAP IOU@0.2","mAP IOU@0.3","mAP IOU@0.4","mAP IOU@0.5","mAP IOU@0.6","mAP IOU@0.7","mAP IOU@0.8","mAP IOU@0.9"],"first_row_in_archive_order":{"model":"TriDet (VideoMAEv2)","paper_title":"Temporal Action Localization with Enhanced Instant Discriminability","paper_url":"/paper/temporal-action-localization-with-enhanced","paper_date":"2023-09-11","arxiv_id":"2309.05590","code_links":[{"title":"dingfengshi/tridet","url":"https://github.com/dingfengshi/tridet"},{"title":"sssste/tridet","url":"https://github.com/sssste/tridet"},{"title":"dingfengshi/tridetplus","url":"https://github.com/dingfengshi/tridetplus"}],"syntology":null}},{"leaderboard":"/sota/temporal-action-localization-on-crosstask","slug":"temporal-action-localization-on-crosstask","dataset":"CrossTask","dataset_url":"/dataset/crosstask","rows_in_archive":7,"metrics":["Recall"],"first_row_in_archive_order":{"model":"VideoCLIP","paper_title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","paper_url":"/paper/videoclip-contrastive-pre-training-for-zero","paper_date":"2021-09-28","arxiv_id":"2109.14084","code_links":[{"title":"facebookresearch/fairseq","url":"https://github.com/facebookresearch/fairseq"},{"title":"pytorch/fairseq","url":"https://github.com/pytorch/fairseq"}],"syntology":null}},{"leaderboard":"/sota/temporal-action-localization-on-epic-kitchens","slug":"temporal-action-localization-on-epic-kitchens","dataset":"EPIC-KITCHENS-100","dataset_url":"/dataset/epic-kitchens-100","rows_in_archive":6,"metrics":["Avg mAP (0.1-0.5)","mAP IOU@0.1","mAP IOU@0.2","mAP IOU@0.3","mAP IOU@0.4","mAP IOU@0.5"],"first_row_in_archive_order":{"model":"AdaTAD (verb, VideoMAE-L)","paper_title":"End-to-End Temporal Action Detection with 1B Parameters Across 1000 Frames","paper_url":"/paper/end-to-end-temporal-action-detection-with-1b","paper_date":"2023-11-28","arxiv_id":"2311.17241","code_links":[{"title":"sming256/OpenTAD","url":"https://github.com/sming256/OpenTAD"},{"title":"sming256/AdaTAD","url":"https://github.com/sming256/AdaTAD"}],"syntology":null}},{"leaderboard":"/sota/temporal-action-localization-on-muses","slug":"temporal-action-localization-on-muses","dataset":"MUSES","dataset_url":"/dataset/muses","rows_in_archive":2,"metrics":["mAP","mAP@0.3","mAP@0.4","mAP@0.5","mAP@0.6","mAP@0.7"],"first_row_in_archive_order":{"model":"TemporalMaxer","paper_title":"TemporalMaxer: Maximize Temporal Context with only Max Pooling for Temporal Action Localization","paper_url":"/paper/temporalmaxer-maximize-temporal-context-with","paper_date":"2023-03-16","arxiv_id":"2303.09055","code_links":[{"title":"tuantng/temporalmaxer","url":"https://github.com/tuantng/temporalmaxer"}],"syntology":null}},{"leaderboard":"/sota/temporal-action-localization-on-activitynet-1","slug":"temporal-action-localization-on-activitynet-1","dataset":"ActivityNet-1.2","dataset_url":"/dataset/activitynet","rows_in_archive":1,"metrics":["mAP IOU@0.5","mAP IOU@0.1","mAP IOU@0.3","mAP IOU@0.7"],"first_row_in_archive_order":{"model":"DeepMetricLearner","paper_title":"Weakly Supervised Temporal Action Localization Using Deep Metric Learning","paper_url":"/paper/weakly-supervised-temporal-action-1","paper_date":"2020-01-21","arxiv_id":"2001.07793","code_links":[{"title":"asrafulashiq/wsad","url":"https://github.com/asrafulashiq/wsad"}],"syntology":null}},{"leaderboard":"/sota/temporal-action-localization-on-ego4d-mq-test","slug":"temporal-action-localization-on-ego4d-mq-test","dataset":"Ego4D MQ test","dataset_url":"/dataset/ego4d","rows_in_archive":1,"metrics":["Average mAP","Recall@1x (tIoU=0.5)"],"first_row_in_archive_order":{"model":"ActionFormer (SlowFast+Omnivore+EgoVLP)","paper_title":"Where a Strong Backbone Meets Strong Features -- ActionFormer for Ego4D Moment Queries Challenge","paper_url":"/paper/where-a-strong-backbone-meets-strong-features","paper_date":"2022-11-16","arxiv_id":"2211.09074","code_links":[{"title":"happyharrycn/actionformer_release","url":"https://github.com/happyharrycn/actionformer_release"},{"title":"showlab/egovlp","url":"https://github.com/showlab/egovlp"}],"syntology":null}},{"leaderboard":"/sota/temporal-action-localization-on-ego4d-mq-val","slug":"temporal-action-localization-on-ego4d-mq-val","dataset":"Ego4D MQ val","dataset_url":"/dataset/ego4d","rows_in_archive":1,"metrics":["Average mAP","Recall@1x (tIoU=0.5)"],"first_row_in_archive_order":{"model":"ActionFormer (SlowFast+Omnivore+EgoVLP)","paper_title":"Where a Strong Backbone Meets Strong Features -- ActionFormer for Ego4D Moment Queries Challenge","paper_url":"/paper/where-a-strong-backbone-meets-strong-features","paper_date":"2022-11-16","arxiv_id":"2211.09074","code_links":[{"title":"happyharrycn/actionformer_release","url":"https://github.com/happyharrycn/actionformer_release"},{"title":"showlab/egovlp","url":"https://github.com/showlab/egovlp"}],"syntology":null}},{"leaderboard":"/sota/temporal-action-localization-on-mexaction2","slug":"temporal-action-localization-on-mexaction2","dataset":"MEXaction2","dataset_url":null,"rows_in_archive":1,"metrics":["mAP"],"first_row_in_archive_order":{"model":"S-CNN","paper_title":"Temporal Action Localization in Untrimmed Videos via Multi-stage CNNs","paper_url":"/paper/temporal-action-localization-in-untrimmed","paper_date":"2016-01-09","arxiv_id":"1601.02129","code_links":[{"title":"zhengshou/scnn","url":"https://github.com/zhengshou/scnn"}],"syntology":null}},{"leaderboard":"/sota/temporal-action-localization-on-thumos-14","slug":"temporal-action-localization-on-thumos-14","dataset":"THUMOS'14","dataset_url":"/dataset/thumos14-1","rows_in_archive":1,"metrics":["mAP IOU@0.5"],"first_row_in_archive_order":{"model":"AVFusion","paper_title":"Hear Me Out: Fusional Approaches for Audio Augmented Temporal Action Localization","paper_url":"/paper/hear-me-out-fusional-approaches-for-audio","paper_date":"2021-06-27","arxiv_id":"2106.14118","code_links":[{"title":"skelemoa/tal-hmo","url":"https://github.com/skelemoa/tal-hmo"}],"syntology":null}},{"leaderboard":"/sota/temporal-action-localization-on-thumos14-2","slug":"temporal-action-localization-on-thumos14-2","dataset":"THUMOS14","dataset_url":"/dataset/thumos14-1","rows_in_archive":1,"metrics":["Avg mAP (0.3:0.7)"],"first_row_in_archive_order":{"model":"BasicTAD (R50-SlowOnly)","paper_title":"BasicTAD: an Astounding RGB-Only Baseline for Temporal Action Detection","paper_url":"/paper/basictad-an-astounding-rgb-only-baseline-for","paper_date":"2022-05-05","arxiv_id":"2205.02717","code_links":[{"title":"mcg-nju/basictad","url":"https://github.com/mcg-nju/basictad"},{"title":"cg1177/dcan","url":"https://github.com/cg1177/dcan"}],"syntology":null}}],"datasets":[{"url":"/dataset/ucf101","name":"UCF101","full_name":"UCF101 Human Actions dataset","num_papers_in_archive":1863},{"url":"/dataset/kinetics","name":"Kinetics","full_name":"Kinetics Human Action Video Dataset","num_papers_in_archive":1341},{"url":"/dataset/hmdb51","name":"HMDB51","full_name":"","num_papers_in_archive":839},{"url":"/dataset/activitynet","name":"ActivityNet","full_name":"","num_papers_in_archive":807},{"url":"/dataset/mpii","name":"MPII","full_name":"MPII Human Pose","num_papers_in_archive":495},{"url":"/dataset/charades","name":"Charades","full_name":"","num_papers_in_archive":428},{"url":"/dataset/thumos14-1","name":"THUMOS14","full_name":"","num_papers_in_archive":318},{"url":"/dataset/kth","name":"KTH","full_name":"KTH Action dataset","num_papers_in_archive":279},{"url":"/dataset/epic-kitchens-100","name":"EPIC-KITCHENS-100","full_name":"","num_papers_in_archive":162},{"url":"/dataset/coin","name":"COIN","full_name":"","num_papers_in_archive":105},{"url":"/dataset/kinetics-700","name":"Kinetics-700","full_name":"Kinetics-700","num_papers_in_archive":95},{"url":"/dataset/finegym","name":"FineGym","full_name":"FineGym","num_papers_in_archive":76},{"url":"/dataset/hacs","name":"HACS","full_name":"Human Action Clips and Segments","num_papers_in_archive":75},{"url":"/dataset/babel-1","name":"BABEL","full_name":"","num_papers_in_archive":72},{"url":"/dataset/utd-mhad","name":"UTD-MHAD","full_name":"","num_papers_in_archive":62},{"url":"/dataset/multithumos","name":"MultiTHUMOS","full_name":"","num_papers_in_archive":58},{"url":"/dataset/crosstask","name":"CrossTask","full_name":"CrossTask","num_papers_in_archive":54},{"url":"/dataset/ego4d","name":"Ego4D","full_name":"","num_papers_in_archive":32},{"url":"/dataset/ikea-asm","name":"IKEA ASM","full_name":"","num_papers_in_archive":25},{"url":"/dataset/fineaction","name":"FineAction","full_name":"","num_papers_in_archive":24},{"url":"/dataset/hieve","name":"HiEve","full_name":"Human-in-Events","num_papers_in_archive":19},{"url":"/dataset/florence3d","name":"Florence3D","full_name":"","num_papers_in_archive":18},{"url":"/dataset/hvu","name":"HVU","full_name":"Holistic Video Understanding","num_papers_in_archive":16},{"url":"/dataset/muses","name":"MUSES","full_name":"MUlti-Shot EventS","num_papers_in_archive":12},{"url":"/dataset/perception-test","name":"Perception Test","full_name":"","num_papers_in_archive":10},{"url":"/dataset/eyth","name":"EYTH","full_name":"EgoYouTubeHands","num_papers_in_archive":7},{"url":"/dataset/msr-actionpairs","name":"MSR ActionPairs","full_name":"","num_papers_in_archive":7},{"url":"/dataset/tum-kitchen","name":"TUM Kitchen","full_name":"TUM Kitchen","num_papers_in_archive":7},{"url":"/dataset/mpii-cooking-2-dataset","name":"MPII Cooking 2 Dataset","full_name":"","num_papers_in_archive":5},{"url":"/dataset/wear","name":"WEAR","full_name":"WEAR: An Outdoor Sports Dataset for Wearable and Egocentric Activity Recognition","num_papers_in_archive":5},{"url":"/dataset/tinyvirat","name":"TinyVIRAT","full_name":"","num_papers_in_archive":4},{"url":"/dataset/composable-activities-dataset","name":"Composable activities dataset","full_name":"","num_papers_in_archive":3},{"url":"/dataset/converse","name":"CONVERSE","full_name":"","num_papers_in_archive":3},{"url":"/dataset/decade","name":"DECADE","full_name":"","num_papers_in_archive":3},{"url":"/dataset/hollywood-3d-dataset","name":"Hollywood 3D dataset","full_name":"","num_papers_in_archive":3},{"url":"/dataset/oreba","name":"OREBA","full_name":"Objectively Recognizing Eating Behavior and Associated Intake","num_papers_in_archive":3},{"url":"/dataset/uav-gesture","name":"UAV-GESTURE","full_name":"","num_papers_in_archive":3},{"url":"/dataset/mcad","name":"MCAD","full_name":"Multi-Camera Action Dataset","num_papers_in_archive":2},{"url":"/dataset/rise","name":"RISE","full_name":null,"num_papers_in_archive":2},{"url":"/dataset/skeletics-152-1","name":"Skeletics 152","full_name":"","num_papers_in_archive":2},{"url":"/dataset/metaphorics","name":"Metaphorics","full_name":"","num_papers_in_archive":1},{"url":"/dataset/whenact","name":"WhenAct","full_name":"Temporal Human Action Localization in Lifestyle Vlogs","num_papers_in_archive":1}],"subtasks":[{"url":"/task/3d-human-action-recognition","name":"3D Action Recognition"},{"url":"/task/action-recognition-in-still-images","name":"Action Recognition In Still Images"},{"url":"/task/activity-recognition-in-videos","name":"Activity Recognition In Videos"},{"url":"/task/open-vocab-temporal-action-detection","name":"Open-vocab Temporal Action Detection"},{"url":"/task/temporal-action-proposal-generation","name":"Temporal Action Proposal Generation"},{"url":"/task/temporal-group-activity-localization","name":"Temporal Group Activity Localization"},{"url":"/task/weakly-supervised-action-localization","name":"Weakly Supervised Action Localization"},{"url":"/task/weakly-supervised-temporal-action","name":"Weakly-supervised Temporal Action Localization"}],"parent_tasks":[{"url":"/task/action-localization","name":"Action Localization"},{"url":"/task/video","name":"Video"},{"url":"/task/zero-shot-learning","name":"Zero-Shot Learning"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":493,"tagged_in_all":1477,"items":[{"url":"/paper/spatial-temporal-graph-convolutional-networks-1","title":"Spatial Temporal Graph Convolutional Networks for Skeleton-Based Action Recognition","date":"2018-01-23","arxiv_id":"1801.07455","repositories_listed":24,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/a-closer-look-at-spatiotemporal-convolutions","title":"A Closer Look at Spatiotemporal Convolutions for Action Recognition","date":"2017-11-30","arxiv_id":"1711.11248","repositories_listed":24,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":4}},{"url":"/paper/grad-cam-improved-visual-explanations-for","title":"Grad-CAM++: Improved Visual Explanations for Deep Convolutional Networks","date":"2017-10-30","arxiv_id":"1710.11063","repositories_listed":24,"syntology":{"n":7,"n_ran":4,"n_unverified":3,"n_pointer_only":3}},{"url":"/paper/temporal-segment-networks-towards-good","title":"Temporal Segment Networks: Towards Good Practices for Deep Action Recognition","date":"2016-08-02","arxiv_id":"1608.00859","repositories_listed":22,"syntology":{"n":24,"n_ran":2,"n_unverified":22,"n_pointer_only":3}},{"url":"/paper/bsn-boundary-sensitive-network-for-temporal","title":"BSN: Boundary Sensitive Network for Temporal Action Proposal Generation","date":"2018-06-08","arxiv_id":"1806.02964","repositories_listed":17,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":4}},{"url":"/paper/bmn-boundary-matching-network-for-temporal","title":"BMN: Boundary-Matching Network for Temporal Action Proposal Generation","date":"2019-07-23","arxiv_id":"1907.09702","repositories_listed":15,"syntology":{"n":11,"n_ran":3,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/unsupervised-learning-of-video","title":"Unsupervised Learning of Video Representations using LSTMs","date":"2015-02-16","arxiv_id":"1502.04681","repositories_listed":12,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/temporal-segment-networks-for-action","title":"Temporal Segment Networks for Action Recognition in Videos","date":"2017-05-08","arxiv_id":"1705.02953","repositories_listed":11,"syntology":null},{"url":"/paper/graph-based-global-reasoning-networks","title":"Graph-Based Global Reasoning Networks","date":"2018-11-30","arxiv_id":"1811.12814","repositories_listed":9,"syntology":{"n":15,"n_ran":8,"n_unverified":7,"n_pointer_only":4}},{"url":"/paper/ava-a-video-dataset-of-spatio-temporally","title":"AVA: A Video Dataset of Spatio-temporally Localized Atomic Visual Actions","date":"2017-05-23","arxiv_id":"1705.08421","repositories_listed":9,"syntology":null},{"url":"/paper/stnet-local-and-global-spatial-temporal","title":"StNet: Local and Global Spatial-Temporal Modeling for Action Recognition","date":"2018-11-05","arxiv_id":"1811.01549","repositories_listed":8,"syntology":null},{"url":"/paper/g-tad-sub-graph-localization-for-temporal","title":"G-TAD: Sub-Graph Localization for Temporal Action Detection","date":"2019-11-26","arxiv_id":"1911.11462","repositories_listed":7,"syntology":{"n":18,"n_ran":3,"n_unverified":15,"n_pointer_only":0}},{"url":"/paper/multivariate-lstm-fcns-for-time-series","title":"Multivariate LSTM-FCNs for Time Series Classification","date":"2018-01-14","arxiv_id":"1801.04503","repositories_listed":7,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/two-stream-convolutional-networks-for-action","title":"Two-Stream Convolutional Networks for Action Recognition in Videos","date":"2014-06-09","arxiv_id":"1406.2199","repositories_listed":7,"syntology":{"n":7,"n_ran":1,"n_unverified":6,"n_pointer_only":2}},{"url":"/paper/ucf101-a-dataset-of-101-human-actions-classes","title":"UCF101: A Dataset of 101 Human Actions Classes From Videos in The Wild","date":"2012-12-03","arxiv_id":"1212.0402","repositories_listed":7,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/eva-exploring-the-limits-of-masked-visual","title":"EVA: Exploring the Limits of Masked Visual Representation Learning at Scale","date":"2022-11-14","arxiv_id":"2211.07636","repositories_listed":6,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/co-occurrence-feature-learning-from-skeleton","title":"Co-occurrence Feature Learning from Skeleton Data for Action Recognition and Detection with Hierarchical Aggregation","date":"2018-04-17","arxiv_id":"1804.06055","repositories_listed":6,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/delving-deeper-into-convolutional-networks","title":"Delving Deeper into Convolutional Networks for Learning Video Representations","date":"2015-11-19","arxiv_id":"1511.06432","repositories_listed":6,"syntology":null},{"url":"/paper/acgnet-action-complement-graph-network-for","title":"ACGNet: Action Complement Graph Network for Weakly-supervised Temporal Action Localization","date":"2021-12-21","arxiv_id":"2112.10977","repositories_listed":5,"syntology":null},{"url":"/paper/vatt-transformers-for-multimodal-self","title":"VATT: Transformers for Multimodal Self-Supervised Learning from Raw Video, Audio and Text","date":"2021-04-22","arxiv_id":"2104.11178","repositories_listed":5,"syntology":{"n":8,"n_ran":5,"n_unverified":3,"n_pointer_only":8}},{"url":"/paper/what-and-how-well-you-performed-a-multitask","title":"What and How Well You Performed? A Multitask Learning Approach to Action Quality Assessment","date":"2019-04-08","arxiv_id":"1904.04346","repositories_listed":5,"syntology":null},{"url":"/paper/representation-flow-for-action-recognition","title":"Representation Flow for Action Recognition","date":"2018-10-02","arxiv_id":"1810.01455","repositories_listed":5,"syntology":null},{"url":"/paper/ridiculously-fast-shot-boundary-detection","title":"Ridiculously Fast Shot Boundary Detection with Fully Convolutional Neural Networks","date":"2017-05-23","arxiv_id":"1705.08214","repositories_listed":5,"syntology":null},{"url":"/paper/towards-good-practices-for-very-deep-two","title":"Towards Good Practices for Very Deep Two-Stream ConvNets","date":"2015-07-08","arxiv_id":"1507.02159","repositories_listed":5,"syntology":null},{"url":"/paper/describing-videos-by-exploiting-temporal","title":"Describing Videos by Exploiting Temporal Structure","date":"2015-02-27","arxiv_id":"1502.08029","repositories_listed":5,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/action-transformer-a-self-attention-model-for","title":"Action Transformer: A Self-Attention Model for Short-Time Pose-Based Human Action Recognition","date":"2021-07-01","arxiv_id":"2107.00606","repositories_listed":4,"syntology":null},{"url":"/paper/non-local-graph-convolutional-networks-for","title":"Two-Stream Adaptive Graph Convolutional Networks for Skeleton-Based Action Recognition","date":"2018-05-20","arxiv_id":"1805.07694","repositories_listed":4,"syntology":null},{"url":"/paper/moments-in-time-dataset-one-million-videos","title":"Moments in Time Dataset: one million videos for event understanding","date":"2018-01-09","arxiv_id":"1801.03150","repositories_listed":4,"syntology":null},{"url":"/paper/im2flow-motion-hallucination-from-static","title":"Im2Flow: Motion Hallucination from Static Images for Action Recognition","date":"2017-12-12","arxiv_id":"1712.04109","repositories_listed":4,"syntology":null},{"url":"/paper/ts-lstm-and-temporal-inception-exploiting","title":"TS-LSTM and Temporal-Inception: Exploiting Spatiotemporal Dynamics for Activity Recognition","date":"2017-03-30","arxiv_id":"1703.10667","repositories_listed":4,"syntology":null}],"syntology_records":16,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}