{"url":"/dataset/movienet","name":"MovieNet","full_name":"MovieNet","description_markdown":"**MovieNet** is a holistic dataset for movie understanding. MovieNet contains 1,100 movies with a large amount of multi-modal data, e.g. trailers, photos, plot descriptions, etc.. Besides, different aspects of manual annotations are provided in MovieNet, including 1.1M characters with bounding boxes and identities, 42K scene boundaries, 2.5K aligned description sentences, 65K tags of place and action, and 92 K tags of cinematic style.","description_withheld":null,"homepage":"https://movienet.github.io/","introduced_date":"2020-07-21","introduced_date_note":null,"introduced_by":{"paper":"/paper/movienet-a-holistic-dataset-for-movie","title":"MovieNet: A Holistic Dataset for Movie Understanding","first_author":"Qingqiu Huang","url":null},"license":null,"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Video Understanding","url":"/task/video-understanding","datasets_with_task":"/datasets/task/video-understanding"},{"name":"Scene Segmentation","url":"/task/scene-segmentation","datasets_with_task":"/datasets/task/scene-segmentation"},{"name":"Audio Generation","url":"/task/audio-generation","datasets_with_task":"/datasets/task/audio-generation"},{"name":"Person Search","url":"/task/person-search","datasets_with_task":"/datasets/task/person-search"}],"languages":[],"variants":["MovieNet"],"data_loaders":[],"num_papers_in_archive":54,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/scene-segmentation-on-movienet","task":"Scene Segmentation","dataset_variant":"MovieNet","rows":2,"metrics":["AP"],"first_row_in_archive_order":{"model":"NeighborNet","paper":"/paper/neighbor-relations-matter-in-video-scene","metrics":{"AP":"71.9"},"code_links":[{"title":"exmorgan-alter/neighbornet","url":"https://github.com/exmorgan-alter/neighbornet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/neighbor-relations-matter-in-video-scene","title":"Neighbor Relations Matter in Video Scene Detection","date":"2024-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/efficient-movie-scene-detection-using-state","title":"Efficient Movie Scene Detection using State-Space Transformers","date":"2022-12-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}