{"url":"/dataset/voxceleb1","name":"VoxCeleb1","full_name":"VoxCeleb1","description_markdown":"**VoxCeleb1** is an audio dataset containing over 100,000 utterances for 1,251 celebrities, extracted from videos uploaded to YouTube.","description_withheld":null,"homepage":"https://www.robots.ox.ac.uk/~vgg/data/voxceleb/vox1.html","introduced_date":"2017-06-26","introduced_date_note":null,"introduced_by":{"paper":"/paper/voxceleb-a-large-scale-speaker-identification","title":"VoxCeleb: a large-scale speaker identification dataset","first_author":null,"url":null},"license":null,"modalities":[{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"task","url":null,"datasets_with_task":"/datasets/task/task"},{"name":"Learning with noisy labels","url":"/task/learning-with-noisy-labels","datasets_with_task":"/datasets/task/learning-with-noisy-labels"},{"name":"Speaker Verification","url":"/task/speaker-verification","datasets_with_task":"/datasets/task/speaker-verification"},{"name":"Talking Head Generation","url":"/task/talking-head-generation","datasets_with_task":"/datasets/task/talking-head-generation"},{"name":"Few-Shot Audio Classification","url":"/task/few-shot-audio-classification","datasets_with_task":"/datasets/task/few-shot-audio-classification"},{"name":"Video Reconstruction","url":"/task/video-reconstruction","datasets_with_task":"/datasets/task/video-reconstruction"},{"name":"Speaker Identification","url":"/task/speaker-identification","datasets_with_task":"/datasets/task/speaker-identification"},{"name":"Speaker Recognition","url":"/task/speaker-recognition","datasets_with_task":"/datasets/task/speaker-recognition"}],"languages":[],"variants":["VoxCeleb","VoxCeleb1","VoxCeleb1 - 1-shot learning","VoxCeleb1 - 8-shot learning","VoxCeleb1 - 32-shot learning"],"data_loaders":[{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/voxceleb","frameworks":["tf","jax"]}],"num_papers_in_archive":680,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/speaker-verification-on-voxceleb","task":"Speaker Verification","dataset_variant":"VoxCeleb","rows":21,"metrics":["EER"],"first_row_in_archive_order":{"model":"SimAM-ResNet100","paper":"/paper/voxblink2-a-100k-speaker-recognition-corpus","metrics":{"EER":"0.20"},"code_links":[{"title":"wenet-e2e/wespeaker","url":"https://github.com/wenet-e2e/wespeaker"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speaker-verification-on-voxceleb1","task":"Speaker Verification","dataset_variant":"VoxCeleb1","rows":16,"metrics":["EER"],"first_row_in_archive_order":{"model":"ReDimNet-B6-SF2-LM-ASNorm (15.0M)","paper":"/paper/reshape-dimensions-network-for-speaker-1","metrics":{"EER":"0.37"},"code_links":[{"title":"IDRnD/ReDimNet","url":"https://github.com/IDRnD/ReDimNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speaker-identification-on-voxceleb1","task":"Speaker Identification","dataset_variant":"VoxCeleb1","rows":12,"metrics":["Top-1 (%)","Top-5 (%)","Number of Params","Accuracy"],"first_row_in_archive_order":{"model":"MSM-MAE","paper":"/paper/masked-modeling-duo-towards-a-universal-audio","metrics":{"Accuracy":"96.6","Top-1 (%)":"96.6"},"code_links":[{"title":"nttcslab/m2d","url":"https://github.com/nttcslab/m2d"},{"title":"nttcslab/eval-audio-repr","url":"https://github.com/nttcslab/eval-audio-repr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/few-shot-audio-classification-on-voxceleb1","task":"Few-Shot Audio Classification","dataset_variant":"VoxCeleb1","rows":10,"metrics":["Top-1 Accuracy(5-Way-1-Shot)"],"first_row_in_archive_order":{"model":"Meta-Curvature (CRNN)","paper":"/paper/metaaudio-a-few-shot-audio-classification","metrics":{"Top-1 Accuracy(5-Way-1-Shot)":"63.85 +- 0.44"},"code_links":[{"title":"cheggan/metaaudio-a-few-shot-audio-classification-benchmark","url":"https://github.com/cheggan/metaaudio-a-few-shot-audio-classification-benchmark"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speaker-recognition-on-voxceleb1","task":"Speaker Recognition","dataset_variant":"VoxCeleb1","rows":2,"metrics":["EER"],"first_row_in_archive_order":{"model":"WavLM+ECAPA-TDNN","paper":"/paper/espnet-spk-full-pipeline-speaker-embedding","metrics":{"EER":"0.39"},"code_links":[{"title":"espnet/espnet","url":"https://github.com/espnet/espnet"},{"title":"Jungjee/RawNet","url":"https://github.com/Jungjee/RawNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/talking-head-generation-on-voxceleb1-1-shot","task":"Talking Head Generation","dataset_variant":"VoxCeleb1 - 1-shot learning","rows":2,"metrics":["FID"],"first_row_in_archive_order":{"model":"Few-shot Adversarial Model","paper":"/paper/few-shot-adversarial-learning-of-realistic","metrics":{"FID":"43.0"},"code_links":[{"title":"vincent-thevenin/Realistic-Neural-Talking-Head-Models","url":"https://github.com/vincent-thevenin/Realistic-Neural-Talking-Head-Models"},{"title":"grey-eye/talking-heads","url":"https://github.com/grey-eye/talking-heads"},{"title":"shoutOutYangJie/Few-Shot-Adversarial-Learning-for-face-swap","url":"https://github.com/shoutOutYangJie/Few-Shot-Adversarial-Learning-for-face-swap"},{"title":"ZVK/Talking-Heads","url":"https://github.com/ZVK/Talking-Heads"},{"title":"ZVK/talking_heads","url":"https://github.com/ZVK/talking_heads"},{"title":"Ierezell/PapierFewShot","url":"https://github.com/Ierezell/PapierFewShot"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/talking-head-generation-on-voxceleb1-32-shot","task":"Talking Head Generation","dataset_variant":"VoxCeleb1 - 32-shot learning","rows":2,"metrics":["FID"],"first_row_in_archive_order":{"model":"Few-shot Adversarial Model","paper":"/paper/few-shot-adversarial-learning-of-realistic","metrics":{"FID":"29.5"},"code_links":[{"title":"vincent-thevenin/Realistic-Neural-Talking-Head-Models","url":"https://github.com/vincent-thevenin/Realistic-Neural-Talking-Head-Models"},{"title":"grey-eye/talking-heads","url":"https://github.com/grey-eye/talking-heads"},{"title":"shoutOutYangJie/Few-Shot-Adversarial-Learning-for-face-swap","url":"https://github.com/shoutOutYangJie/Few-Shot-Adversarial-Learning-for-face-swap"},{"title":"ZVK/Talking-Heads","url":"https://github.com/ZVK/Talking-Heads"},{"title":"ZVK/talking_heads","url":"https://github.com/ZVK/talking_heads"},{"title":"Ierezell/PapierFewShot","url":"https://github.com/Ierezell/PapierFewShot"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/talking-head-generation-on-voxceleb1-8-shot","task":"Talking Head Generation","dataset_variant":"VoxCeleb1 - 8-shot learning","rows":2,"metrics":["FID"],"first_row_in_archive_order":{"model":"Few-shot Adversarial Model","paper":"/paper/few-shot-adversarial-learning-of-realistic","metrics":{"FID":"38.0"},"code_links":[{"title":"vincent-thevenin/Realistic-Neural-Talking-Head-Models","url":"https://github.com/vincent-thevenin/Realistic-Neural-Talking-Head-Models"},{"title":"grey-eye/talking-heads","url":"https://github.com/grey-eye/talking-heads"},{"title":"shoutOutYangJie/Few-Shot-Adversarial-Learning-for-face-swap","url":"https://github.com/shoutOutYangJie/Few-Shot-Adversarial-Learning-for-face-swap"},{"title":"ZVK/Talking-Heads","url":"https://github.com/ZVK/Talking-Heads"},{"title":"ZVK/talking_heads","url":"https://github.com/ZVK/talking_heads"},{"title":"Ierezell/PapierFewShot","url":"https://github.com/Ierezell/PapierFewShot"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-reconstruction-on-voxceleb","task":"Video Reconstruction","dataset_variant":"VoxCeleb","rows":2,"metrics":["AED","AKD","L1"],"first_row_in_archive_order":{"model":"Siarohin et al.","paper":"/paper/motion-representations-for-articulated-1","metrics":{"AED":"0.133","AKD":"1.28","L1":"0.040"},"code_links":[{"title":"AliaksandrSiarohin/first-order-model","url":"https://github.com/AliaksandrSiarohin/first-order-model"},{"title":"snap-research/articulated-animation","url":"https://github.com/snap-research/articulated-animation"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/on-voxceleb1","task":"","dataset_variant":"VoxCeleb1","rows":1,"metrics":["Acc"],"first_row_in_archive_order":{"model":"M2D/0.7","paper":"/paper/masked-modeling-duo-towards-a-universal-audio","metrics":{"Acc":"96.3"},"code_links":[{"title":"nttcslab/m2d","url":"https://github.com/nttcslab/m2d"},{"title":"nttcslab/eval-audio-repr","url":"https://github.com/nttcslab/eval-audio-repr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/reshape-dimensions-network-for-speaker-1","title":"Reshape Dimensions Network for Speaker Recognition","date":"2024-07-25","rows_on_this_dataset":28,"code_links":1,"syntology":null},{"paper":"/paper/voxblink2-a-100k-speaker-recognition-corpus","title":"VoxBlink2: A 100K+ Speaker Recognition Corpus and the Open-Set Speaker-Identification Benchmark","date":"2024-07-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/ssamba-self-supervised-audio-representation","title":"SSAMBA: Self-Supervised Audio Representation Learning with Mamba State Space Model","date":"2024-05-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":5,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/masked-modeling-duo-towards-a-universal-audio","title":"Masked Modeling Duo: Towards a Universal Audio Pre-training Framework","date":"2024-04-09","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":7,"samples_unverified":0,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/espnet-spk-full-pipeline-speaker-embedding","title":"ESPnet-SPK: full pipeline speaker embedding toolkit with reproducible recipes, self-supervised front-ends, and off-the-shelf models","date":"2024-01-30","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":9,"samples_unverified":6,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mt-slvr-multi-task-self-supervised-learning","title":"MT-SLVR: Multi-Task Self-Supervised Learning for Transformation In(Variant) Representations","date":"2023-05-29","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/masked-modeling-duo-learning-representations","title":"Masked Modeling Duo: Learning Representations by Encouraging Both Networks to Model the Input","date":"2022-10-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/masked-autoencoders-that-listen","title":"Masked Autoencoders that Listen","date":"2022-07-13","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":32,"samples_ran":17,"samples_unverified":15,"pointer_only_for_licence":10,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/atst-audio-representation-learning-with","title":"ATST: Audio Representation Learning with Teacher-Student Transformer","date":"2022-04-26","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/metaaudio-a-few-shot-audio-classification","title":"MetaAudio: A Few-Shot Audio Classification Benchmark","date":"2022-04-05","rows_on_this_dataset":7,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-fine-tuned-wav2vec-2-0-hubert-benchmark-for","title":"A Fine-tuned Wav2vec 2.0/HuBERT Benchmark For Speech Emotion Recognition, Speaker Verification and Spoken Language Understanding","date":"2021-11-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/ssast-self-supervised-audio-spectrogram","title":"SSAST: Self-Supervised Audio Spectrogram Transformer","date":"2021-10-19","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":11,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/titanet-neural-model-for-speaker","title":"TitaNet: Neural Model for speaker representation with 1D Depth-wise separable convolutions and global context","date":"2021-10-08","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-task-voice-activated-framework-using","title":"Multi-task Voice Activated Framework using Self-supervised Learning","date":"2021-10-03","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/fine-tuning-wav2vec2-for-speaker-recognition","title":"Fine-tuning wav2vec2 for speaker recognition","date":"2021-09-30","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/speechnas-towards-better-trade-off-between","title":"SpeechNAS: Towards Better Trade-off between Latency and Accuracy for Large-Scale Speaker Verification","date":"2021-09-18","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/motion-representations-for-articulated-1","title":"Motion Representations for Articulated Animation","date":"2021-04-22","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":2,"samples_unverified":4,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/contrastive-learning-of-general-purpose-audio","title":"Contrastive Learning of General-Purpose Audio Representations","date":"2020-10-21","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/autospeech-neural-architecture-search-for","title":"AutoSpeech: Neural Architecture Search for Speaker Recognition","date":"2020-05-07","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/few-shot-adversarial-learning-of-realistic","title":"Few-Shot Adversarial Learning of Realistic Neural Talking Head Models","date":"2019-05-20","rows_on_this_dataset":3,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/x2face-a-network-for-controlling-face-1","title":"X2Face: A network for controlling face generation using images, audio, and pose codes","date":"2018-09-01","rows_on_this_dataset":3,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":10,"samples_harvested":94,"samples_ran":57,"samples_unverified":37,"pointer_only_for_licence":30,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}