{"url":"/task/video-understanding","name":"Video Understanding","slug":"video-understanding","description_markdown":"A crucial task of **Video Understanding** is to recognise and localise (in space and time) different actions or events appearing in the video.\r\n\r\n\r\n<span class=\"description-source\">Source: [Action Detection from a Robot-Car Perspective ](https://arxiv.org/abs/1807.11332)</span>","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":1149,"papers_with_code":542,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":56,"subtasks":7,"parent_tasks":1},"benchmarks":[],"datasets":[{"url":"/dataset/kinetics","name":"Kinetics","full_name":"Kinetics Human Action Video Dataset","num_papers_in_archive":1341},{"url":"/dataset/charades","name":"Charades","full_name":"","num_papers_in_archive":428},{"url":"/dataset/charades-sta","name":"Charades-STA","full_name":"","num_papers_in_archive":236},{"url":"/dataset/shanghaitech-campus","name":"ShanghaiTech Campus","full_name":"","num_papers_in_archive":207},{"url":"/dataset/seed-bench","name":"SEED-Bench","full_name":"","num_papers_in_archive":137},{"url":"/dataset/ava","name":"AVA","full_name":"Atomic Visual Actions","num_papers_in_archive":113},{"url":"/dataset/soccernet-v2","name":"SoccerNet-v2","full_name":"","num_papers_in_archive":58},{"url":"/dataset/movienet","name":"MovieNet","full_name":"MovieNet","num_papers_in_archive":54},{"url":"/dataset/internvid","name":"InternVid","full_name":"","num_papers_in_archive":46},{"url":"/dataset/epic-kitchens","name":"EPIC-KITCHENS-55","full_name":"","num_papers_in_archive":42},{"url":"/dataset/charades-ego","name":"Charades-Ego","full_name":"","num_papers_in_archive":33},{"url":"/dataset/mtl-aqa","name":"MTL-AQA","full_name":"","num_papers_in_archive":33},{"url":"/dataset/ccd","name":"CCD","full_name":"Car Crash Dataset","num_papers_in_archive":18},{"url":"/dataset/vidsitu","name":"VidSitu","full_name":"","num_papers_in_archive":18},{"url":"/dataset/situated-reasoning-star","name":"STAR Benchmark","full_name":"Situated Reasoning","num_papers_in_archive":17},{"url":"/dataset/hvu","name":"HVU","full_name":"Holistic Video Understanding","num_papers_in_archive":16},{"url":"/dataset/queryd","name":"QuerYD","full_name":"","num_papers_in_archive":15},{"url":"/dataset/ovbench","name":"OVBench","full_name":"","num_papers_in_archive":14},{"url":"/dataset/aesthetic-visual-analysis","name":"Aesthetic Visual Analysis","full_name":"Aesthetic Visual Analysis","num_papers_in_archive":12},{"url":"/dataset/dramaqa","name":"DramaQA","full_name":"","num_papers_in_archive":12},{"url":"/dataset/imagecode","name":"ImageCoDe","full_name":"Image Retrieval from Contextual Descriptions","num_papers_in_archive":11},{"url":"/dataset/moviescope","name":"Moviescope","full_name":"","num_papers_in_archive":6},{"url":"/dataset/v2c","name":"V2C","full_name":"Video-to-Commonsense","num_papers_in_archive":6},{"url":"/dataset/chronomagic","name":"ChronoMagic","full_name":"","num_papers_in_archive":5},{"url":"/dataset/infinibench","name":"InfiniBench","full_name":"InfiniBench: A Comprehensive Benchmark for Large Multimodal Models in Very Long Video Understanding","num_papers_in_archive":5},{"url":"/dataset/mlb-youtube-dataset","name":"MLB-YouTube Dataset","full_name":null,"num_papers_in_archive":5},{"url":"/dataset/real-life-violence-situations-dataset","name":"Real Life Violence Situations Dataset","full_name":"","num_papers_in_archive":5},{"url":"/dataset/soccerdb","name":"SoccerDB","full_name":"","num_papers_in_archive":5},{"url":"/dataset/m-3-vos","name":"M$^3$-VOS","full_name":"M$^3$-VOS: Multi-Phase, Multi-Transition, and Multi-Scenery Video Object Segmentation","num_papers_in_archive":4},{"url":"/dataset/query-focused-video-summarization-dataset","name":"Query-Focused Video Summarization Dataset","full_name":"","num_papers_in_archive":4},{"url":"/dataset/converse","name":"CONVERSE","full_name":"","num_papers_in_archive":3},{"url":"/dataset/deepsportradar-v1","name":"DeepSportRadar-v1","full_name":"","num_papers_in_archive":3},{"url":"/dataset/fitness-aqa","name":"Fitness-AQA","full_name":"Fitness Action Quality Assessment [ECCV 2022]","num_papers_in_archive":3},{"url":"/dataset/3massiv","name":"3MASSIV","full_name":"","num_papers_in_archive":2},{"url":"/dataset/chronomagic-pro","name":"ChronoMagic-Pro","full_name":"","num_papers_in_archive":2},{"url":"/dataset/cinescale","name":"Cinescale","full_name":"CineScale: A dataset of cinematic shot scale in movies","num_papers_in_archive":2},{"url":"/dataset/egok360","name":"EGOK360","full_name":"","num_papers_in_archive":2},{"url":"/dataset/stanford-ecm","name":"Stanford-ECM","full_name":"Stanford-ECM","num_papers_in_archive":2},{"url":"/dataset/svbench","name":"SVBench","full_name":"Streaming Video Understanding Benchmark","num_papers_in_archive":2},{"url":"/dataset/vtc","name":"VTC","full_name":"Videos, Titles and Comments","num_papers_in_archive":2},{"url":"/dataset/airletters","name":"AirLetters","full_name":"","num_papers_in_archive":1},{"url":"/dataset/c3d-features-for-phd2gif","name":"C3D features for PHD2GIF","full_name":"","num_papers_in_archive":1},{"url":"/dataset/chronomagic-proh","name":"ChronoMagic-ProH","full_name":"","num_papers_in_archive":1},{"url":"/dataset/cinepile","name":"CinePile: A Long Video Question Answering Dataset and Benchmark","full_name":"","num_papers_in_archive":1},{"url":"/dataset/custom-spatio-temporal-action-video-dataset","name":"Custom Spatio-Temporal Action Video Dataset","full_name":"","num_papers_in_archive":1},{"url":"/dataset/i3-video","name":"i3-video","full_name":"is-it-instructional-video","num_papers_in_archive":1},{"url":"/dataset/kinetic-gebc","name":"Kinetics-GEB+","full_name":"","num_papers_in_archive":1},{"url":"/dataset/lsdbench","name":"LSDBench","full_name":"Long-video Sampling Dilemma Benchmark","num_papers_in_archive":1},{"url":"/dataset/musechat-dataset","name":"MuseChat Dataset","full_name":"MuseChat: A Conversational Music Recommendation System for Videos (CVPR 2024 Highlight Paper)","num_papers_in_archive":1},{"url":"/dataset/placepedia","name":"Placepedia","full_name":"Placepedia","num_papers_in_archive":1},{"url":"/dataset/soccernet-echoes-1","name":"SoccerNet-Echoes","full_name":"SoccerNet-Echoes: A Soccer Game Audio Commentary Dataset","num_papers_in_archive":1},{"url":"/dataset/trailers12k","name":"Trailers12k","full_name":"","num_papers_in_archive":1},{"url":"/dataset/vcg-112k","name":"VCG+112K","full_name":"Video Instruction Dataset 112K","num_papers_in_archive":1},{"url":"/dataset/wildqa","name":"WildQA","full_name":"","num_papers_in_archive":1},{"url":"/dataset/win-fail-action-understanding","name":"Win-Fail Action Understanding","full_name":"Win-Fail Action Understanding","num_papers_in_archive":1},{"url":"/dataset/vript","name":"Vript","full_name":"🎬 Vript: A Video Is Worth Thousands of Words","num_papers_in_archive":0}],"subtasks":[{"url":"/task/anomaly-detection-in-surveillance-videos","name":"Anomaly Detection In Surveillance Videos"},{"url":"/task/causal-discovery-in-video-reasoning","name":"Causal Discovery in Video Reasoning"},{"url":"/task/long-video-activity-recognition","name":"Long-video Activity Recognition"},{"url":"/task/streaming-video-understanding","name":"Streaming video understanding"},{"url":"/task/temporal-sentence-grounding","name":"Temporal Sentence Grounding"},{"url":"/task/video-alignment","name":"Video Alignment"},{"url":"/task/video-quality-assessment","name":"Video Quality Assessment"}],"parent_tasks":[{"url":"/task/video","name":"Video"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":542,"tagged_in_all":1149,"items":[{"url":"/paper/is-space-time-attention-all-you-need-for","title":"Is Space-Time Attention All You Need for Video Understanding?","date":"2021-02-09","arxiv_id":"2102.05095","repositories_listed":16,"syntology":{"n":43,"n_ran":35,"n_unverified":8,"n_pointer_only":14}},{"url":"/paper/video-swin-transformer","title":"Video Swin Transformer","date":"2021-06-24","arxiv_id":"2106.13230","repositories_listed":15,"syntology":{"n":32,"n_ran":7,"n_unverified":25,"n_pointer_only":0}},{"url":"/paper/temporal-shift-module-for-efficient-video","title":"TSM: Temporal Shift Module for Efficient Video Understanding","date":"2018-11-20","arxiv_id":"1811.08383","repositories_listed":13,"syntology":{"n":16,"n_ran":6,"n_unverified":10,"n_pointer_only":4}},{"url":"/paper/tokenlearner-what-can-8-learned-tokens-do-for","title":"TokenLearner: What Can 8 Learned Tokens Do for Images and Videos?","date":"2021-06-21","arxiv_id":"2106.11297","repositories_listed":11,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/videomae-masked-autoencoders-are-data-1","title":"VideoMAE: Masked Autoencoders are Data-Efficient Learners for Self-Supervised Video Pre-Training","date":"2022-03-23","arxiv_id":"2203.12602","repositories_listed":9,"syntology":{"n":13,"n_ran":9,"n_unverified":4,"n_pointer_only":12}},{"url":"/paper/ava-a-video-dataset-of-spatio-temporally","title":"AVA: A Video Dataset of Spatio-temporally Localized Atomic Visual Actions","date":"2017-05-23","arxiv_id":"1705.08421","repositories_listed":9,"syntology":null},{"url":"/paper/soccernet-2022-challenges-results","title":"SoccerNet 2022 Challenges Results","date":"2022-10-05","arxiv_id":"2210.02365","repositories_listed":7,"syntology":null},{"url":"/paper/video-instance-segmentation","title":"Video Instance Segmentation","date":"2019-05-12","arxiv_id":"1905.04804","repositories_listed":6,"syntology":{"n":11,"n_ran":3,"n_unverified":8,"n_pointer_only":1}},{"url":"/paper/eco-efficient-convolutional-network-for","title":"ECO: Efficient Convolutional Network for Online Video Understanding","date":"2018-04-24","arxiv_id":"1804.09066","repositories_listed":6,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/flamingo-a-visual-language-model-for-few-shot-1","title":"Flamingo: a Visual Language Model for Few-Shot Learning","date":"2022-04-29","arxiv_id":"2204.14198","repositories_listed":5,"syntology":{"n":24,"n_ran":18,"n_unverified":6,"n_pointer_only":7}},{"url":"/paper/clip4clip-an-empirical-study-of-clip-for-end","title":"CLIP4Clip: An Empirical Study of CLIP for End to End Video Clip Retrieval","date":"2021-04-18","arxiv_id":"2104.08860","repositories_listed":5,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/representation-flow-for-action-recognition","title":"Representation Flow for Action Recognition","date":"2018-10-02","arxiv_id":"1810.01455","repositories_listed":5,"syntology":null},{"url":"/paper/learnable-pooling-with-context-gating-for","title":"Learnable pooling with Context Gating for video classification","date":"2017-06-21","arxiv_id":"1706.06905","repositories_listed":5,"syntology":null},{"url":"/paper/chat-univi-unified-visual-representation","title":"Chat-UniVi: Unified Visual Representation Empowers Large Language Models with Image and Video Understanding","date":"2023-11-14","arxiv_id":"2311.08046","repositories_listed":4,"syntology":null},{"url":"/paper/video-llama-an-instruction-tuned-audio-visual","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","date":"2023-06-05","arxiv_id":"2306.02858","repositories_listed":4,"syntology":{"n":25,"n_ran":18,"n_unverified":7,"n_pointer_only":9}},{"url":"/paper/deepsportradar-v1-computer-vision-dataset-for","title":"DeepSportradar-v1: Computer Vision Dataset for Sports Understanding with High Quality Annotations","date":"2022-08-17","arxiv_id":"2208.08190","repositories_listed":4,"syntology":null},{"url":"/paper/cvnets-high-performance-library-for-computer","title":"CVNets: High Performance Library for Computer Vision","date":"2022-06-04","arxiv_id":"2206.02002","repositories_listed":4,"syntology":null},{"url":"/paper/tsm-temporal-shift-module-for-efficient-and","title":"TSM: Temporal Shift Module for Efficient and Scalable Video Understanding on Edge Device","date":"2021-09-27","arxiv_id":"2109.13227","repositories_listed":4,"syntology":null},{"url":"/paper/soccernet-v2-a-dataset-and-benchmarks-for","title":"SoccerNet-v2: A Dataset and Benchmarks for Holistic Understanding of Broadcast Soccer Videos","date":"2020-11-26","arxiv_id":"2011.13367","repositories_listed":4,"syntology":null},{"url":"/paper/temporal-interlacing-network","title":"Temporal Interlacing Network","date":"2020-01-17","arxiv_id":"2001.06499","repositories_listed":4,"syntology":{"n":25,"n_ran":12,"n_unverified":13,"n_pointer_only":0}},{"url":"/paper/long-term-feature-banks-for-detailed-video","title":"Long-Term Feature Banks for Detailed Video Understanding","date":"2018-12-12","arxiv_id":"1812.05038","repositories_listed":4,"syntology":{"n":8,"n_ran":0,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/virtualhome-simulating-household-activities","title":"VirtualHome: Simulating Household Activities via Programs","date":"2018-06-19","arxiv_id":"1806.07011","repositories_listed":4,"syntology":null},{"url":"/paper/ts-lstm-and-temporal-inception-exploiting","title":"TS-LSTM and Temporal-Inception: Exploiting Spatiotemporal Dynamics for Activity Recognition","date":"2017-03-30","arxiv_id":"1703.10667","repositories_listed":4,"syntology":null},{"url":"/paper/cogvlm2-visual-language-models-for-image-and","title":"CogVLM2: Visual Language Models for Image and Video Understanding","date":"2024-08-29","arxiv_id":"2408.16500","repositories_listed":3,"syntology":{"n":17,"n_ran":12,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/mlvu-a-comprehensive-benchmark-for-multi-task","title":"MLVU: Benchmarking Multi-task Long Video Understanding","date":"2024-06-06","arxiv_id":"2406.04264","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/no-time-to-waste-squeeze-time-into-channel","title":"No Time to Waste: Squeeze Time into Channel for Mobile Video Understanding","date":"2024-05-14","arxiv_id":"2405.08344","repositories_listed":3,"syntology":null},{"url":"/paper/panoptic-video-scene-graph-generation-1","title":"Panoptic Video Scene Graph Generation","date":"2023-11-28","arxiv_id":"2311.17058","repositories_listed":3,"syntology":{"n":11,"n_ran":11,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/mvbench-a-comprehensive-multi-modal-video","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","date":"2023-11-28","arxiv_id":"2311.17005","repositories_listed":3,"syntology":{"n":10,"n_ran":7,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/uniformerv2-spatiotemporal-learning-by-arming-1","title":"UniFormerV2: Spatiotemporal Learning by Arming Image ViTs with Video UniFormer","date":"2022-11-17","arxiv_id":"2211.09552","repositories_listed":3,"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/temporal-action-segmentation-an-analysis-of","title":"Temporal Action Segmentation: An Analysis of Modern Techniques","date":"2022-10-19","arxiv_id":"2210.10352","repositories_listed":3,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":0}}],"syntology_records":18,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}