{"url":"/dataset/easycom","name":"EasyCom","full_name":null,"description_markdown":"The Easy Communications (EasyCom) dataset is a world-first dataset designed to help mitigate the cocktail party effect from an augmented-reality (AR) -motivated multi-sensor egocentric world view. The dataset contains AR glasses egocentric multi-channel microphone array audio, wide field-of-view RGB video, speech source pose, headset microphone audio, annotated voice activity, speech transcriptions, head and face bounding boxes and source identification labels. We have created and are releasing this dataset to facilitate research in multi-modal AR solutions to the cocktail party problem.\r\n\r\nSource: [EasyCom](https://github.com/facebookresearch/EasyComDataset)","description_withheld":null,"homepage":"https://github.com/facebookresearch/EasyComDataset","introduced_date":"2021-07-09","introduced_date_note":null,"introduced_by":{"paper":"/paper/easycom-an-augmented-reality-dataset-to","title":"EasyCom: An Augmented Reality Dataset to Support Algorithms for Easy Communication in Noisy Environments","first_author":"Jacob Donley","url":null},"license":{"name":"CC BY-NC 4.0","url":"https://creativecommons.org/licenses/by-nc/4.0/"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Audio","url":"/datasets/modality/audio"},{"name":"Time series","url":"/datasets/modality/time-series"},{"name":"Speech","url":"/datasets/modality/speech"},{"name":"Dialog","url":"/datasets/modality/dialog"},{"name":"RGB Video","url":"/datasets/modality/rgb-video"}],"tasks":[{"name":"Speech Recognition","url":"/task/speech-recognition","datasets_with_task":"/datasets/task/speech-recognition"},{"name":"Speech Enhancement","url":"/task/speech-enhancement","datasets_with_task":"/datasets/task/speech-enhancement"},{"name":"Face Clustering","url":"/task/face-clustering","datasets_with_task":"/datasets/task/face-clustering"},{"name":"Active Speaker Localization","url":"/task/active-speaker-localization","datasets_with_task":"/datasets/task/active-speaker-localization"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["EasyCom"],"data_loaders":[],"num_papers_in_archive":22,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/speech-enhancement-on-easycom","task":"Speech Enhancement","dataset_variant":"EasyCom","rows":6,"metrics":["PESQ","STOI","ViSQOL","HASQI","Audio Quality MOS","SDR","ESTOI","HASPI","SI-SDR","SIIB","SNR","SegSNR"],"first_row_in_archive_order":{"model":"MaxDI (Baseline)","paper":"/paper/easycom-an-augmented-reality-dataset-to","metrics":{"ESTOI":"0.379","HASPI":"0.830","HASQI":"0.249","PESQ":"1.17","SDR":"-12.9","SI-SDR":"-23.4","SIIB":"139","SNR":"-10.1","STOI":"0.544","SegSNR":"-12.2","ViSQOL":"1.68"},"code_links":[{"title":"facebookresearch/EasyComDataset","url":"https://github.com/facebookresearch/EasyComDataset"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-recognition-on-easycom","task":"Speech Recognition","dataset_variant":"EasyCom","rows":5,"metrics":["WER (%)"],"first_row_in_archive_order":{"model":"ReVISE (bf)","paper":"/paper/revise-self-supervised-speech-resynthesis","metrics":{"WER (%)":"52.1"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/active-speaker-localization-on-easycom","task":"Active Speaker Localization","dataset_variant":"EasyCom","rows":1,"metrics":["ASL mAP"],"first_row_in_archive_order":{"model":"AV (cor+eng+box)","paper":"/paper/egocentric-deep-multi-channel-audio-visual","metrics":{"ASL mAP":"0.8632"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/face-clustering-on-easycom","task":"Face Clustering","dataset_variant":"EasyCom","rows":1,"metrics":["NMI"],"first_row_in_archive_order":{"model":"VC TRSF.","paper":"/paper/self-supervised-video-centralised-transformer","metrics":{"NMI":"87.44"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/revise-self-supervised-speech-resynthesis","title":"ReVISE: Self-Supervised Speech Resynthesis with Visual Input for Universal and Generalized Speech Enhancement","date":"2022-12-21","rows_on_this_dataset":8,"code_links":0,"syntology":null},{"paper":"/paper/direction-aware-joint-adaptation-of-neural","title":"Direction-Aware Joint Adaptation of Neural Speech Enhancement and Recognition in Real Multiparty Conversational Environments","date":"2022-07-15","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/self-supervised-video-centralised-transformer","title":"Self-supervised Video-centralised Transformer for Video Face Clustering","date":"2022-03-24","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/egocentric-deep-multi-channel-audio-visual","title":"Egocentric Deep Multi-Channel Audio-Visual Active Speaker Localization","date":"2022-01-06","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/easycom-an-augmented-reality-dataset-to","title":"EasyCom: An Augmented Reality Dataset to Support Algorithms for Easy Communication in Noisy Environments","date":"2021-07-09","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}