{"url":"/dataset/avspeech","name":"AVSpeech","full_name":null,"description_markdown":"**AVSpeech** is a large-scale audio-visual dataset comprising\r\nspeech clips with no interfering background signals. The segments\r\nare of varying length, between 3 and 10 seconds long, and in each clip\r\nthe only visible face in the video and audible sound in the soundtrack\r\nbelong to a single speaking person. In total, the dataset contains\r\nroughly 4700 hours of video segments with approximately 150,000\r\ndistinct speakers, spanning a wide variety of people, languages\r\nand face poses.","description_withheld":null,"homepage":"https://looking-to-listen.github.io/avspeech/","introduced_date":"2018-04-10","introduced_date_note":null,"introduced_by":{"paper":"/paper/looking-to-listen-at-the-cocktail-party-a","title":"Looking to Listen at the Cocktail Party: A Speaker-Independent Audio-Visual Model for Speech Separation","first_author":"Ariel Ephrat","url":null},"license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Audio","url":"/datasets/modality/audio"},{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Audio Source Separation","url":"/task/audio-source-separation","datasets_with_task":"/datasets/task/audio-source-separation"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"French","url":"/datasets/language/french"},{"name":"Spanish","url":"/datasets/language/spanish"},{"name":"German","url":"/datasets/language/german"},{"name":"Italian","url":"/datasets/language/italian"},{"name":"Chinese","url":"/datasets/language/chinese"},{"name":"Multilingual","url":"/datasets/language/multilingual"},{"name":"Japanese","url":"/datasets/language/japanese"},{"name":"Russian","url":"/datasets/language/russian"},{"name":"Portuguese","url":"/datasets/language/portuguese"},{"name":"Arabic","url":"/datasets/language/arabic"},{"name":"Korean","url":"/datasets/language/korean"}],"variants":["AVSpeech"],"data_loaders":[],"num_papers_in_archive":42,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}