{"url":"/dataset/a2d","name":"A2D","full_name":"Actor-Action Dataset","description_markdown":"A2D (Actor-Action Dataset) is a dataset for simultaneously inferring actors and actions in videos. A2D has seven actor classes (adult, baby, ball, bird, car, cat, and dog) and eight action classes (climb, crawl, eat, fly, jump, roll, run, and walk) not including the no-action class, which we also consider. The A2D has 3,782 videos with at least 99 instances per valid actor-action tuple and videos are labeled with both pixel-level actors and actions for sampled frames. The A2D dataset serves as a large-scale testbed for various vision problems: video-level single- and multiple-label actor-action recognition, instance-level object segmentation/co-segmentation, as well as pixel-level actor-action semantic segmentation to name a few.","description_withheld":null,"homepage":"https://web.eecs.umich.edu/~jjcorso/r/a2d/","introduced_date":"2015-06-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/can-humans-fly-action-understanding-with","title":"Can Humans Fly? Action Understanding With Multiple Classes of Actors","first_author":"Chenliang Xu","url":null},"license":{"name":"Custom (non-commercial)","url":"https://web.eecs.umich.edu/~jjcorso/r/a2d/files/README"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Object Detection","url":"/task/object-detection","datasets_with_task":"/datasets/task/object-detection"}],"languages":[],"variants":["A2D"],"data_loaders":[],"num_papers_in_archive":42,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/object-detection-on-a2d","task":"Object Detection","dataset_variant":"A2D","rows":1,"metrics":["Mean IoU"],"first_row_in_archive_order":{"model":"RL [10] Lpixel","paper":"/paper/paint-transformer-feed-forward-neural","metrics":{"Mean IoU":"5.8"},"code_links":[{"title":"huage001/painttransformer","url":"https://github.com/huage001/painttransformer"},{"title":"wzmsltw/painttransformer","url":"https://github.com/wzmsltw/painttransformer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/paint-transformer-feed-forward-neural","title":"Paint Transformer: Feed Forward Neural Painting with Stroke Prediction","date":"2021-08-09","rows_on_this_dataset":1,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}