{"url":"/dataset/okutama-action","name":"Okutama-Action","full_name":null,"description_markdown":"A new video dataset for aerial view concurrent human action detection. It consists of 43 minute-long fully-annotated sequences with 12 action classes. Okutama-Action features many challenges missing in current datasets, including dynamic transition of actions, significant changes in scale and aspect ratio, abrupt camera movement, as well as multi-labeled actors.\r\n\r\nSource: [Okutama-Action: An Aerial View Video Dataset for Concurrent Human Action Detection](/paper/okutama-action-an-aerial-view-video-dataset)","description_withheld":null,"homepage":"http://okutama-action.org","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/okutama-action-an-aerial-view-video-dataset","title":"Okutama-Action: An Aerial View Video Dataset for Concurrent Human Action Detection","first_author":"Mohammadamin Barekatain","url":null},"license":null,"modalities":[],"tasks":[{"name":"Action Recognition","url":"/task/action-recognition-in-videos","datasets_with_task":"/datasets/task/action-recognition-in-videos"},{"name":"Object Tracking","url":"/task/object-tracking","datasets_with_task":"/datasets/task/object-tracking"},{"name":"Action Detection","url":"/task/action-detection","datasets_with_task":"/datasets/task/action-detection"}],"languages":[],"variants":["Okutama-Action"],"data_loaders":[],"num_papers_in_archive":24,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/action-recognition-on-okutama-action","task":"Action Recognition","dataset_variant":"Okutama-Action","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PLAR with bbox (Ours)","paper":"/paper/prompt-learning-for-action-recognition","metrics":{"Accuracy":"75.93"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/prompt-learning-for-action-recognition","title":"SCP: Soft Conditional Prompt Learning for Aerial Video Action Recognition","date":"2023-05-21","rows_on_this_dataset":2,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}