{"url":"/dataset/egtea","name":"EGTEA","full_name":"EGTEA Gaze+","description_markdown":"Extended GTEA Gaze+\r\nEGTEA Gaze+ is a large-scale dataset for FPV actions and gaze. It subsumes GTEA Gaze+ and comes with HD videos (1280x960), audios, gaze tracking data, frame-level action annotations, and pixel-level hand masks at sampled frames.\r\nSpecifically, EGTEA Gaze+ contains 28 hours (de-identified) of cooking activities from 86 unique sessions of 32 subjects. These videos come with audios and gaze tracking (30Hz). We have further provided human annotations of actions (human-object interactions) and hand masks.\r\n\r\nThe action annotations include 10325 instances of fine-grained actions, such as \"Cut bell pepper\" or \"Pour condiment (from) condiment container into salad\".\r\n\r\nThe hand annotations consist of 15,176 hand masks from 13,847 frames from the videos.\r\n\r\nSource: [http://cbs.ic.gatech.edu/fpv/](http://cbs.ic.gatech.edu/fpv/)\r\nImage Source: [http://cbs.ic.gatech.edu/fpv/](http://cbs.ic.gatech.edu/fpv/)","description_withheld":null,"homepage":"http://cbs.ic.gatech.edu/fpv/","introduced_date":"2018-09-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/in-the-eye-of-beholder-joint-learning-of-gaze","title":"In the Eye of Beholder: Joint Learning of Gaze and Actions in First Person Video","first_author":"Yin Li","url":null},"license":null,"modalities":[],"tasks":[{"name":"Long-tail Learning","url":"/task/long-tail-learning","datasets_with_task":"/datasets/task/long-tail-learning"},{"name":"Action Anticipation","url":"/task/action-anticipation","datasets_with_task":"/datasets/task/action-anticipation"},{"name":"Egocentric Activity Recognition","url":"/task/egocentric-activity-recognition","datasets_with_task":"/datasets/task/egocentric-activity-recognition"}],"languages":[],"variants":[],"data_loaders":[],"num_papers_in_archive":100,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/egocentric-activity-recognition-on-egtea-1","task":"Egocentric Activity Recognition","dataset_variant":"EGTEA","rows":6,"metrics":["Average Accuracy","Mean class accuracy"],"first_row_in_archive_order":{"model":"LaViLa (Finetuned, TimeSformer-L)","paper":"/paper/learning-video-representations-from-large","metrics":{"Average Accuracy":"81.75","Mean class accuracy":"76"},"code_links":[{"title":"facebookresearch/lavila","url":"https://github.com/facebookresearch/lavila"},{"title":"Ziyang412/VideoTree","url":"https://github.com/Ziyang412/VideoTree"},{"title":"ceezh/llovi","url":"https://github.com/ceezh/llovi"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-anticipation-on-egtea","task":"Action Anticipation","dataset_variant":"EGTEA","rows":3,"metrics":["Top-1 Accuracy"],"first_row_in_archive_order":{"model":"UADT","paper":"/paper/uncertainty-aware-action-decoupling","metrics":{"Top-1 Accuracy":"68.4"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/long-tail-learning-on-egtea","task":"Long-tail Learning","dataset_variant":"EGTEA","rows":3,"metrics":["Average Precision","Average Recall"],"first_row_in_archive_order":{"model":"CDB-loss (3D- ResNeXt101)","paper":"/paper/class-wise-difficulty-balanced-loss-for","metrics":{"Average Precision":"63.86","Average Recall":"66.24"},"code_links":[{"title":"hitachi-rd-cv/CDB-loss","url":"https://github.com/hitachi-rd-cv/CDB-loss"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/uncertainty-aware-action-decoupling","title":"Uncertainty-aware Action Decoupling Transformer for Action Anticipation","date":"2024-01-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/learning-video-representations-from-large","title":"Learning Video Representations from Large Language Models","date":"2022-12-08","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":20,"samples_ran":6,"samples_unverified":14,"pointer_only_for_licence":20,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/interaction-visual-transformer-for-egocentric","title":"Interaction Region Visual Transformer for Egocentric Action Anticipation","date":"2022-11-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/predicting-the-next-action-by-modeling-the","title":"Predicting the Next Action by Modeling the Abstract Goal","date":"2022-09-12","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/group-contextualization-for-video-recognition","title":"Group Contextualization for Video Recognition","date":"2022-03-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":5,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/integrating-human-gaze-into-attention-for","title":"Integrating Human Gaze into Attention for Egocentric Activity Recognition","date":"2020-11-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/class-wise-difficulty-balanced-loss-for","title":"Class-Wise Difficulty-Balanced Loss for Solving Class-Imbalance","date":"2020-10-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/symbiotic-attention-with-privileged","title":"Symbiotic Attention with Privileged Information for Egocentric Action Recognition","date":"2020-02-08","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/class-balanced-loss-based-on-effective-number","title":"Class-Balanced Loss Based on Effective Number of Samples","date":"2019-01-16","rows_on_this_dataset":1,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":27,"samples_ran":7,"samples_unverified":20,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lsta-long-short-term-attention-for-egocentric","title":"LSTA: Long Short-Term Attention for Egocentric Action Recognition","date":"2018-11-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/attention-is-all-we-need-nailing-down-object","title":"Attention is All We Need: Nailing Down Object-centric Attention for Egocentric Activity Recognition","date":"2018-07-31","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/focal-loss-for-dense-object-detection","title":"Focal Loss for Dense Object Detection","date":"2017-08-07","rows_on_this_dataset":1,"code_links":234,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":11,"samples_unverified":0,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":5,"samples_harvested":65,"samples_ran":29,"samples_unverified":36,"pointer_only_for_licence":28,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}