{"url":"/dataset/arca23k","name":"ARCA23K","full_name":null,"description_markdown":"ARCA23K is a dataset of labelled sound events created to investigate real-world label noise. It contains 23,727 audio clips originating from Freesound, and each clip belongs to one of 70 classes taken from the AudioSet ontology. The dataset was created using an entirely automated process with no manual verification of the data. For this reason, many clips are expected to be labelled incorrectly.","description_withheld":null,"homepage":"https://zenodo.org/record/5117901","introduced_date":"2021-09-19","introduced_date_note":null,"introduced_by":{"paper":"/paper/arca23k-an-audio-dataset-for-investigating","title":"ARCA23K: An audio dataset for investigating open-set label noise","first_author":"Turab Iqbal","url":null},"license":null,"modalities":[{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[],"languages":[],"variants":["ARCA23K"],"data_loaders":[],"num_papers_in_archive":2,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}