{"url":"/task/action-anticipation","name":"Action Anticipation","slug":"action-anticipation","description_markdown":"Next action anticipation is defined as observing 1, ... , T frames and predicting the action that happens after a gap of T_a seconds. It is important to note that a new action starts after T_a seconds that is not seen in the observed frames. Here T_a=1 second.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":110,"papers_with_code":49,"benchmarks":8,"benchmark_tables_in_archive":8,"benchmark_tables_shown":8,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":11,"subtasks":0,"parent_tasks":2},"benchmarks":[{"leaderboard":"/sota/action-anticipation-on-epic-kitchens-100","slug":"action-anticipation-on-epic-kitchens-100","dataset":"EPIC-KITCHENS-100","dataset_url":"/dataset/epic-kitchens-100","rows_in_archive":9,"metrics":["Recall@5","Top-5 Verb","Top-5 Noun"],"first_row_in_archive_order":{"model":"PlausiVL","paper_title":"Can't make an Omelette without Breaking some Eggs: Plausible Action Anticipation using Large Video-Language Models","paper_url":"/paper/can-t-make-an-omelette-without-breaking-some","paper_date":"2024-05-30","arxiv_id":"2405.20305","code_links":[],"syntology":null}},{"leaderboard":"/sota/action-anticipation-on-epic-kitchens-100-test","slug":"action-anticipation-on-epic-kitchens-100-test","dataset":"EPIC-KITCHENS-100 (test)","dataset_url":null,"rows_in_archive":8,"metrics":["recall@5"],"first_row_in_archive_order":{"model":"InAViT","paper_title":"Interaction Region Visual Transformer for Egocentric Action Anticipation","paper_url":"/paper/interaction-visual-transformer-for-egocentric","paper_date":"2022-11-25","arxiv_id":"2211.14154","code_links":[{"title":"lahaproject/inavit","url":"https://github.com/lahaproject/inavit"}],"syntology":null}},{"leaderboard":"/sota/action-anticipation-on-epic-kitchens-55-1","slug":"action-anticipation-on-epic-kitchens-55-1","dataset":"EPIC-KITCHENS-55 (Unseen test set (S2)","dataset_url":null,"rows_in_archive":7,"metrics":["Top 1 Accuracy - Act.","Top 1 Accuracy - Noun","Top 1 Accuracy - Verb","Top 5 Accuracy - Act.","Top 5 Accuracy - Noun","Top 5 Accuracy - Verb"],"first_row_in_archive_order":{"model":"Abstract Goal","paper_title":"Predicting the Next Action by Modeling the Abstract Goal","paper_url":"/paper/predicting-the-next-action-by-modeling-the","paper_date":"2022-09-12","arxiv_id":"2209.05044","code_links":[],"syntology":null}},{"leaderboard":"/sota/action-anticipation-on-epic-kitchens-55-seen","slug":"action-anticipation-on-epic-kitchens-55-seen","dataset":"EPIC-KITCHENS-55 (Seen test set (S1))","dataset_url":null,"rows_in_archive":7,"metrics":["Top 1 Accuracy - Act.","Top 1 Accuracy - Noun","Top 1 Accuracy - Verb","Top 5 Accuracy - Act.","Top 5 Accuracy - Noun","Top 5 Accuracy - Verb"],"first_row_in_archive_order":{"model":"Abstract Goal","paper_title":"Predicting the Next Action by Modeling the Abstract Goal","paper_url":"/paper/predicting-the-next-action-by-modeling-the","paper_date":"2022-09-12","arxiv_id":"2209.05044","code_links":[],"syntology":null}},{"leaderboard":"/sota/action-anticipation-on-egtea","slug":"action-anticipation-on-egtea","dataset":"EGTEA","dataset_url":"/dataset/egtea","rows_in_archive":3,"metrics":["Top-1 Accuracy"],"first_row_in_archive_order":{"model":"UADT","paper_title":"Uncertainty-aware Action Decoupling Transformer for Action Anticipation","paper_url":"/paper/uncertainty-aware-action-decoupling","paper_date":"2024-01-01","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/action-anticipation-on-assembly101","slug":"action-anticipation-on-assembly101","dataset":"Assembly101","dataset_url":"/dataset/assembly101","rows_in_archive":2,"metrics":["Verbs Recall@5","Objects Recall@5","Actions Recall@5"],"first_row_in_archive_order":{"model":"Goal Consistency","paper_title":"Action Anticipation with Goal Consistency","paper_url":"/paper/action-anticipation-with-goal-consistency","paper_date":"2023-06-26","arxiv_id":"2306.15045","code_links":[{"title":"olga-zats/goal_consistency","url":"https://github.com/olga-zats/goal_consistency"}],"syntology":null}},{"leaderboard":"/sota/action-anticipation-on-egoexolearn","slug":"action-anticipation-on-egoexolearn","dataset":"EgoExoLearn","dataset_url":"/dataset/egoexolearn","rows_in_archive":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Action anticipation baseline (co-training, with gaze)","paper_title":"EgoExoLearn: A Dataset for Bridging Asynchronous Ego- and Exo-centric View of Procedural Activities in Real World","paper_url":"/paper/egoexolearn-a-dataset-for-bridging","paper_date":"2024-03-24","arxiv_id":"2403.16182","code_links":[{"title":"opengvlab/egoexolearn","url":"https://github.com/opengvlab/egoexolearn"}],"syntology":{"n":8,"n_ran":7,"n_unverified":1,"n_pointer_only":0}}},{"leaderboard":"/sota/action-anticipation-on-50-salads","slug":"action-anticipation-on-50-salads","dataset":"50-Salads","dataset_url":null,"rows_in_archive":1,"metrics":["Top-1 Accuracy"],"first_row_in_archive_order":{"model":"UADT","paper_title":"Uncertainty-aware Action Decoupling Transformer for Action Anticipation","paper_url":"/paper/uncertainty-aware-action-decoupling","paper_date":"2024-01-01","arxiv_id":null,"code_links":[],"syntology":null}}],"datasets":[{"url":"/dataset/epic-kitchens-100","name":"EPIC-KITCHENS-100","full_name":"","num_papers_in_archive":162},{"url":"/dataset/egtea","name":"EGTEA","full_name":"EGTEA Gaze+","num_papers_in_archive":100},{"url":"/dataset/assembly101","name":"Assembly101","full_name":"","num_papers_in_archive":57},{"url":"/dataset/ego4d","name":"Ego4D","full_name":"","num_papers_in_archive":32},{"url":"/dataset/tvseries","name":"TVSeries","full_name":"","num_papers_in_archive":29},{"url":"/dataset/egoexolearn","name":"EgoExoLearn","full_name":"","num_papers_in_archive":12},{"url":"/dataset/viena2","name":"VIENA2","full_name":"","num_papers_in_archive":4},{"url":"/dataset/mm-or","name":"MM-OR","full_name":"","num_papers_in_archive":3},{"url":"/dataset/deep-future-gaze","name":"OST","full_name":"Egocentric Dataset","num_papers_in_archive":3},{"url":"/dataset/cp2a-dataset","name":"CP2A dataset","full_name":"CARLA Pedestrian Action Anticipation dataset","num_papers_in_archive":1},{"url":"/dataset/darai","name":"DARai","full_name":"Daily Activity Recordings for AI and ML applications","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/2d-human-pose-estimation","name":"2D Human Pose Estimation"},{"url":"/task/action-recognition-in-videos-2","name":"Action Recognition In Videos"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":49,"tagged_in_all":110,"items":[{"url":"/paper/rescaling-egocentric-vision","title":"Rescaling Egocentric Vision","date":"2020-06-23","arxiv_id":"2006.13256","repositories_listed":7,"syntology":{"n":18,"n_ran":3,"n_unverified":15,"n_pointer_only":0}},{"url":"/paper/temporal-aggregate-representations-for-long","title":"Temporal Aggregate Representations for Long-Range Video Understanding","date":"2020-06-01","arxiv_id":"2006.00830","repositories_listed":2,"syntology":null},{"url":"/paper/rolling-unrolling-lstms-for-action","title":"Rolling-Unrolling LSTMs for Action Anticipation from First-Person Video","date":"2020-05-04","arxiv_id":"2005.02190","repositories_listed":2,"syntology":null},{"url":"/paper/hallucinet-ing-spatiotemporal-representations","title":"HalluciNet-ing Spatiotemporal Representations Using a 2D-CNN","date":"2019-12-10","arxiv_id":"1912.04430","repositories_listed":2,"syntology":null},{"url":"/paper/what-would-you-expect-anticipating-egocentric","title":"What Would You Expect? Anticipating Egocentric Actions with Rolling-Unrolling LSTMs and Modality Attention","date":"2019-05-22","arxiv_id":"1905.09035","repositories_listed":2,"syntology":null},{"url":"/paper/scaling-egocentric-vision-the-epic-kitchens","title":"Scaling Egocentric Vision: The EPIC-KITCHENS Dataset","date":"2018-04-08","arxiv_id":"1804.02748","repositories_listed":2,"syntology":null},{"url":"/paper/v-jepa-2-self-supervised-video-models-enable","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","date":"2025-06-11","arxiv_id":"2506.09985","repositories_listed":1,"syntology":{"n":8,"n_ran":0,"n_unverified":8,"n_pointer_only":8}},{"url":"/paper/hierarchical-and-multimodal-data-for-daily","title":"Hierarchical and Multimodal Data for Daily Activity Understanding","date":"2025-04-24","arxiv_id":"2504.17696","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-large-models-are-effective-action","title":"Multimodal Large Models Are Effective Action Anticipators","date":"2025-01-01","arxiv_id":"2501.00795","repositories_listed":1,"syntology":null},{"url":"/paper/manta-diffusion-mamba-for-efficient-and-1","title":"MANTA: Diffusion Mamba for Efficient and Effective Stochastic Long-Term Dense Action Anticipation","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mamba-fusion-learning-actions-through","title":"Mamba Fusion: Learning Actions Through Questioning","date":"2024-09-17","arxiv_id":"2409.11513","repositories_listed":1,"syntology":null},{"url":"/paper/2408-02769","title":"From Recognition to Prediction: Leveraging Sequence Reasoning for Action Anticipation","date":"2024-08-05","arxiv_id":"2408.02769","repositories_listed":1,"syntology":null},{"url":"/paper/quiil-at-t3-challenge-towards-automation-in","title":"QuIIL at T3 challenge: Towards Automation in Life-Saving Intervention Procedures from First-Person View","date":"2024-07-18","arxiv_id":"2407.13216","repositories_listed":1,"syntology":null},{"url":"/paper/gated-temporal-diffusion-for-stochastic-long","title":"Gated Temporal Diffusion for Stochastic Long-Term Dense Anticipation","date":"2024-07-16","arxiv_id":"2407.11954","repositories_listed":1,"syntology":null},{"url":"/paper/semantically-guided-representation-learning","title":"Semantically Guided Representation Learning For Action Anticipation","date":"2024-07-02","arxiv_id":"2407.02309","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/egovideo-exploring-egocentric-foundation","title":"EgoVideo: Exploring Egocentric Foundation Model and Downstream Adaptation","date":"2024-06-26","arxiv_id":"2406.18070","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_unverified":4,"n_pointer_only":15}},{"url":"/paper/egoexolearn-a-dataset-for-bridging","title":"EgoExoLearn: A Dataset for Bridging Asynchronous Ego- and Exo-centric View of Procedural Activities in Real World","date":"2024-03-24","arxiv_id":"2403.16182","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/action-scene-graphs-for-long-form","title":"Action Scene Graphs for Long-Form Understanding of Egocentric Videos","date":"2023-12-06","arxiv_id":"2312.03391","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/object-centric-video-representation-for-long","title":"Object-centric Video Representation for Long-term Action Anticipation","date":"2023-10-31","arxiv_id":"2311.00180","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/antgpt-can-large-language-models-help-long","title":"AntGPT: Can Large Language Models Help Long-term Action Anticipation from Videos?","date":"2023-07-31","arxiv_id":"2307.16368","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/technical-report-for-ego4d-long-term-action","title":"Technical Report for Ego4D Long Term Action Anticipation Challenge 2023","date":"2023-07-04","arxiv_id":"2307.01467","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/palm-predicting-actions-through-language","title":"Palm: Predicting Actions through Language Models @ Ego4D Long-Term Action Anticipation Challenge 2023","date":"2023-06-28","arxiv_id":"2306.16545","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/action-anticipation-with-goal-consistency","title":"Action Anticipation with Goal Consistency","date":"2023-06-26","arxiv_id":"2306.15045","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-next-active-object-based-egocentric","title":"Enhancing Next Active Object-based Egocentric Action Anticipation with Guided Attention","date":"2023-05-22","arxiv_id":"2305.12953","repositories_listed":1,"syntology":null},{"url":"/paper/fine-grained-affordance-annotation-for","title":"Fine-grained Affordance Annotation for Egocentric Hand-Object Interaction Videos","date":"2023-02-07","arxiv_id":"2302.03292","repositories_listed":1,"syntology":null},{"url":"/paper/interaction-visual-transformer-for-egocentric","title":"Interaction Region Visual Transformer for Egocentric Action Anticipation","date":"2022-11-25","arxiv_id":"2211.14154","repositories_listed":1,"syntology":null},{"url":"/paper/anticipative-feature-fusion-transformer-for","title":"Anticipative Feature Fusion Transformer for Multi-Modal Action Anticipation","date":"2022-10-23","arxiv_id":"2210.12649","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-learning-approaches-for-long-term","title":"Rethinking Learning Approaches for Long-Term Action Anticipation","date":"2022-10-20","arxiv_id":"2210.11566","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/distilling-knowledge-from-language-models-for","title":"Text-Derived Knowledge Helps Vision: A Simple Cross-modal Distillation for Video-based Action Anticipation","date":"2022-10-12","arxiv_id":"2210.05991","repositories_listed":1,"syntology":null},{"url":"/paper/learning-state-aware-visual-representations","title":"Learning State-Aware Visual Representations from Audible Interactions","date":"2022-09-27","arxiv_id":"2209.13583","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}}],"syntology_records":12,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}