{"url":"/dataset/mad","name":"MAD","full_name":null,"description_markdown":"MAD (Movie Audio Descriptions) is an automatically curated large-scale dataset for the task of natural language grounding in videos or natural language moment retrieval.\r\nMAD exploits available audio descriptions of mainstream movies. Such audio descriptions are redacted for visually impaired audiences and are therefore highly descriptive of the visual content being displayed. \r\nMAD contains over 384,000 natural language sentences grounded in over 1,200 hours of video, and provides a unique setup for video grounding as the visual stream is truly untrimmed with an average video duration of 110 minutes. 2 orders of magnitude longer than legacy datasets. \r\n\r\nTake a look at the paper for additional information.\r\n\r\nFrom the authors on availability: \"Due to copyright constraints, MAD’s videos will not be publicly released. However, we will provide all necessary features for our experiments’ reproducibility and promote future research in this direction\"","description_withheld":null,"homepage":"https://github.com/Soldelli/MAD","introduced_date":"2021-12-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/mad-a-scalable-dataset-for-language-grounding","title":"MAD: A Scalable Dataset for Language Grounding in Videos from Movie Audio Descriptions","first_author":"Mattia Soldan","url":null},"license":null,"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Natural Language Moment Retrieval","url":"/task/natural-language-moment-retrieval","datasets_with_task":"/datasets/task/natural-language-moment-retrieval"},{"name":"Moment Retrieval","url":"/task/moment-retrieval","datasets_with_task":"/datasets/task/moment-retrieval"},{"name":"Video Grounding","url":"/task/video-grounding","datasets_with_task":"/datasets/task/video-grounding"},{"name":"Natural Language Visual Grounding","url":"/task/natural-language-visual-grounding","datasets_with_task":"/datasets/task/natural-language-visual-grounding"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["MAD"],"data_loaders":[],"num_papers_in_archive":36,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/natural-language-moment-retrieval-on-mad","task":"Natural Language Moment Retrieval","dataset_variant":"MAD","rows":8,"metrics":["R@1,IoU=0.1","R@1,IoU=0.3","R@1,IoU=0.5","R@10,IoU=0.1","R@10,IoU=0.3","R@10,IoU=0.5","R@100,IoU=0.1","R@100,IoU=0.3","R@100,IoU=0.5","R@5,IoU=0.1","R@5,IoU=0.5","R@50,IoU=0.1","R@50,IoU=0.3","R@50,IoU=0.5","R@5,IoU=0.3"],"first_row_in_archive_order":{"model":"ReVisionLLM","paper":"/paper/revisionllm-recursive-vision-language-model","metrics":{"R@1,IoU=0.1":"17.3","R@1,IoU=0.3":"12.7","R@1,IoU=0.5":"6.7"},"code_links":[{"title":"tanveer81/revisionllm","url":"https://github.com/tanveer81/revisionllm"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-grounding-on-mad","task":"Video Grounding","dataset_variant":"MAD","rows":2,"metrics":["R@1,IoU=0.1","R@5,IoU=0.1","R@10,IoU=0.1","R@100,IoU=0.1","R@50,IoU=0.1","R@1,IoU=0.3","R@5,IoU=0.3"],"first_row_in_archive_order":{"model":"DeCafNet","paper":"/paper/decafnet-delegate-and-conquer-for-efficient","metrics":{"R@1,IoU=0.1":"13.25","R@1,IoU=0.3":"10.96","R@5,IoU=0.1":"27.73","R@5,IoU=0.3":"23.68"},"code_links":[{"title":"zijialewislu/cvpr2025-decafnet","url":"https://github.com/zijialewislu/cvpr2025-decafnet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/decafnet-delegate-and-conquer-for-efficient","title":"DeCafNet: Delegate and Conquer for Efficient Temporal Grounding in Long Videos","date":"2025-05-22","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/revisionllm-recursive-vision-language-model","title":"ReVisionLLM: Recursive Vision-Language Model for Temporal Grounding in Hour-Long Videos","date":"2024-11-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/rgnet-a-unified-retrieval-and-grounding","title":"RGNet: A Unified Clip Retrieval and Grounding Network for Long Videos","date":"2023-12-11","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/boundary-denoising-for-video-activity","title":"Boundary-Denoising for Video Activity Localization","date":"2023-04-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/localizing-moments-in-long-video-via","title":"Localizing Moments in Long Video Via Multimodal Guidance","date":"2023-02-26","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/mad-a-scalable-dataset-for-language-grounding","title":"MAD: A Scalable Dataset for Language Grounding in Videos from Movie Audio Descriptions","date":"2021-12-01","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}