{"url":"/task/action-understanding","name":"Action Understanding","slug":"action-understanding","description_markdown":null,"categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":88,"papers_with_code":35,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":4,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/action-understanding-on-win-fail-action","slug":"action-understanding-on-win-fail-action","dataset":"Win-Fail Action Understanding","dataset_url":"/dataset/win-fail-action-understanding","rows_in_archive":1,"metrics":["2-Class Accuracy"],"first_row_in_archive_order":{"model":"2DCNN+TRN","paper_title":"Win-Fail Action Recognition","paper_url":"/paper/win-fail-action-recognition","paper_date":"2021-02-15","arxiv_id":"2102.07355","code_links":[{"title":"ParitoshParmar/Win-Fail-Action-Recognition","url":"https://github.com/ParitoshParmar/Win-Fail-Action-Recognition"}],"syntology":null}}],"datasets":[{"url":"/dataset/mtl-aqa","name":"MTL-AQA","full_name":"","num_papers_in_archive":33},{"url":"/dataset/fitness-aqa","name":"Fitness-AQA","full_name":"Fitness Action Quality Assessment [ECCV 2022]","num_papers_in_archive":3},{"url":"/dataset/w-oops","name":"W-Oops","full_name":"","num_papers_in_archive":1},{"url":"/dataset/win-fail-action-understanding","name":"Win-Fail Action Understanding","full_name":"Win-Fail Action Understanding","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":35,"tagged_in_all":88,"items":[{"url":"/paper/prompted-contrast-with-masked-motion-modeling","title":"Prompted Contrast with Masked Motion Modeling: Towards Versatile 3D Action Representation Learning","date":"2023-08-08","arxiv_id":"2308.03975","repositories_listed":2,"syntology":null},{"url":"/paper/llava-pose-enhancing-human-pose-and-action","title":"LLaVA-Pose: Enhancing Human Pose and Action Understanding via Keypoint-Integrated Instruction Tuning","date":"2025-06-26","arxiv_id":"2506.21317","repositories_listed":1,"syntology":null},{"url":"/paper/f-3-set-towards-analyzing-fast-frequent-and","title":"F$^3$Set: Towards Analyzing Fast, Frequent, and Fine-grained Events from Videos","date":"2025-04-11","arxiv_id":"2504.08222","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_unverified":3,"n_pointer_only":12}},{"url":"/paper/llavaction-evaluating-and-training-multi","title":"LLaVAction: evaluating and training multi-modal large language models for action recognition","date":"2025-03-24","arxiv_id":"2503.18712","repositories_listed":1,"syntology":null},{"url":"/paper/sefar-semi-supervised-fine-grained-action","title":"SeFAR: Semi-supervised Fine-grained Action Recognition with Temporal Perturbation and Learning Stabilization","date":"2025-01-02","arxiv_id":"2501.01245","repositories_listed":1,"syntology":null},{"url":"/paper/language-assisted-skeleton-action","title":"Language-Assisted Skeleton Action Understanding for Skeleton-Based Temporal Action Segmentation","date":"2024-10-31","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/egoexo-fitness-towards-egocentric-and","title":"EgoExo-Fitness: Towards Egocentric and Exocentric Full-Body Action Understanding","date":"2024-06-13","arxiv_id":"2406.08877","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/ophnet-a-large-scale-video-benchmark-for","title":"OphNet: A Large-Scale Video Benchmark for Ophthalmic Surgical Workflow Understanding","date":"2024-06-11","arxiv_id":"2406.07471","repositories_listed":1,"syntology":{"n":16,"n_ran":10,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/self-supervised-skeleton-action","title":"Self-Supervised Skeleton-Based Action Representation Learning: A Benchmark and Beyond","date":"2024-06-05","arxiv_id":"2406.02978","repositories_listed":1,"syntology":null},{"url":"/paper/fineparser-a-fine-grained-spatio-temporal","title":"FineParser: A Fine-grained Spatio-temporal Action Parser for Human-centric Action Quality Assessment","date":"2024-05-11","arxiv_id":"2405.06887","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":5}},{"url":"/paper/sports-qa-a-large-scale-video-question","title":"Sports-QA: A Large-Scale Video Question Answering Benchmark for Complex and Professional Sports","date":"2024-01-03","arxiv_id":"2401.01505","repositories_listed":1,"syntology":null},{"url":"/paper/finesports-a-multi-person-hierarchical-sports","title":"FineSports: A Multi-person Hierarchical Sports Video Dataset for Fine-grained Action Understanding","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/open-vocabulary-video-relation-extraction","title":"Open-Vocabulary Video Relation Extraction","date":"2023-12-25","arxiv_id":"2312.15670","repositories_listed":1,"syntology":null},{"url":"/paper/unified-multi-modal-unsupervised","title":"Unified Multi-modal Unsupervised Representation Learning for Skeleton-based Action Understanding","date":"2023-11-06","arxiv_id":"2311.03106","repositories_listed":1,"syntology":null},{"url":"/paper/memory-and-anticipation-transformer-for","title":"Memory-and-Anticipation Transformer for Online Action Understanding","date":"2023-08-15","arxiv_id":"2308.07893","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/paxion-patching-action-knowledge-in-video-1","title":"Paxion: Patching Action Knowledge in Video-Language Foundation Models","date":"2023-05-18","arxiv_id":"2305.10683","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/act-thor-a-controlled-benchmark-for-embodied","title":"ACT-Thor: A Controlled Benchmark for Embodied Action Understanding in Simulated Environments","date":"2022-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/weakly-supervised-temporal-action-detection","title":"Weakly-Supervised Temporal Action Detection for Fine-Grained Videos with Hierarchical Atomic Actions","date":"2022-07-24","arxiv_id":"2207.11805","repositories_listed":1,"syntology":null},{"url":"/paper/action-quality-assessment-with-temporal","title":"Action Quality Assessment with Temporal Parsing Transformer","date":"2022-07-19","arxiv_id":"2207.09270","repositories_listed":1,"syntology":null},{"url":"/paper/tragedy-plus-time-capturing-unintended-human","title":"Tragedy Plus Time: Capturing Unintended Human Activities from Weakly-labeled Videos","date":"2022-04-28","arxiv_id":"2204.13548","repositories_listed":1,"syntology":null},{"url":"/paper/bridge-prompt-towards-ordinal-action","title":"Bridge-Prompt: Towards Ordinal Action Understanding in Instructional Videos","date":"2022-03-26","arxiv_id":"2203.14104","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/domain-knowledge-informed-self-supervised","title":"Domain Knowledge-Informed Self-Supervised Representations for Workout Form Assessment","date":"2022-02-28","arxiv_id":"2202.14019","repositories_listed":1,"syntology":null},{"url":"/paper/towards-tokenized-human-dynamics","title":"Towards Tokenized Human Dynamics Representation","date":"2021-11-22","arxiv_id":"2111.11433","repositories_listed":1,"syntology":null},{"url":"/paper/video-pose-distillation-for-few-shot-fine","title":"Video Pose Distillation for Few-Shot, Fine-Grained Sports Action Recognition","date":"2021-09-03","arxiv_id":"2109.01305","repositories_listed":1,"syntology":null},{"url":"/paper/few-shot-fine-grained-action-recognition-via","title":"Few-Shot Fine-Grained Action Recognition via Bidirectional Attention and Contrastive Meta-Learning","date":"2021-08-15","arxiv_id":"2108.06647","repositories_listed":1,"syntology":null},{"url":"/paper/piano-a-parametric-hand-bone-model-from","title":"PIANO: A Parametric Hand Bone Model from Magnetic Resonance Imaging","date":"2021-06-21","arxiv_id":"2106.10893","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_unverified":2,"n_pointer_only":7}},{"url":"/paper/home-action-genome-cooperative-compositional","title":"Home Action Genome: Cooperative Compositional Action Understanding","date":"2021-05-11","arxiv_id":"2105.05226","repositories_listed":1,"syntology":{"n":16,"n_ran":2,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/win-fail-action-recognition","title":"Win-Fail Action Recognition","date":"2021-02-15","arxiv_id":"2102.07355","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-relational-modeling-with-self","title":"Temporal Relational Modeling with Self-Supervision for Action Segmentation","date":"2020-12-14","arxiv_id":"2012.07508","repositories_listed":1,"syntology":null},{"url":"/paper/video-action-understanding-a-tutorial","title":"Video Action Understanding","date":"2020-10-13","arxiv_id":"2010.06647","repositories_listed":1,"syntology":null}],"syntology_records":9,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}