{"url":"/dataset/dipco","name":"DiPCo","full_name":"DiPCo -- Dinner Party Corpus","description_markdown":"We present a speech data corpus that simulates a \"dinner party\" scenario taking place in an everyday home environment. The corpus was created by recording multiple groups of four Amazon employee volunteers having a natural conversation in English around a dining table. The participants were recorded by a single-channel close-talk microphone and by five far-field 7-microphone array devices positioned at different locations in the recording room. The dataset contains the audio recordings and human labeled transcripts of a total of 10 sessions with a duration between 15 and 45 minutes. The corpus was created to advance in the field of noise robust and distant speech processing and is intended to serve as a public research and benchmarking data set.","description_withheld":null,"homepage":"https://zenodo.org/record/8122551","introduced_date":"2019-09-30","introduced_date_note":null,"introduced_by":{"paper":"/paper/dipco-dinner-party-corpus","title":"DiPCo -- Dinner Party Corpus","first_author":"Maarten Van Segbroeck","url":null},"license":{"name":"https://zenodo.org/record/8122551","url":"https://cdla.dev/"},"modalities":[{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Speech Separation","url":"/task/speech-separation","datasets_with_task":"/datasets/task/speech-separation"},{"name":"Distant Speech Recognition","url":"/task/distant-speech-recognition","datasets_with_task":"/datasets/task/distant-speech-recognition"},{"name":"Robust Speech Recognition","url":"/task/robust-speech-recognition","datasets_with_task":"/datasets/task/robust-speech-recognition"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["DiPCo"],"data_loaders":[],"num_papers_in_archive":16,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}