{"url":"/dataset/sounddescs","name":"SoundDescs","full_name":null,"description_markdown":"We introduce a new audio dataset called SoundDescs that can be used for tasks such as text to audio retrieval, audio captioning etc. This dataset contains 32,979 pairs of audio files and text descriptions. There are 23 categories found in SoundDescs including but not limited to nature, clocks, fire etc.\r\n\r\nSoundDescs can be downloaded from [here](https://github.com/akoepke/audio-retrieval-benchmark) and retrieval results for this dataset can be found in the associated paper [Audio Retrieval with Natural Language Queries: A Benchmark Study](https://arxiv.org/pdf/2112.09418.pdf).","description_withheld":null,"homepage":"https://www.robots.ox.ac.uk/~vgg/research/audio-retrieval/","introduced_date":"2021-12-17","introduced_date_note":null,"introduced_by":{"paper":"/paper/audio-retrieval-with-natural-language-queries-1","title":"Audio Retrieval with Natural Language Queries: A Benchmark Study","first_author":"A. Sophia Koepke","url":null},"license":{"name":"https://sound-effects.bbcrewind.co.uk/licensing","url":"https://github.com/akoepke/audio-retrieval-benchmark"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Text to Audio Retrieval","url":"/task/text-to-audio-retrieval","datasets_with_task":"/datasets/task/text-to-audio-retrieval"},{"name":"Audio to Text Retrieval","url":"/task/audio-to-text-retrieval","datasets_with_task":"/datasets/task/audio-to-text-retrieval"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["SoundDescs"],"data_loaders":[],"num_papers_in_archive":8,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/text-to-audio-retrieval-on-sounddescs","task":"Text to Audio Retrieval","dataset_variant":"SoundDescs","rows":4,"metrics":["R@1","R@10"],"first_row_in_archive_order":{"model":"CE","paper":"/paper/audio-retrieval-with-natural-language-queries-1","metrics":{"R@1":"31.1±0.2","R@10":"70.8±0.5"},"code_links":[{"title":"akoepke/audio-retrieval-benchmark","url":"https://github.com/akoepke/audio-retrieval-benchmark"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/audio-retrieval-with-natural-language-queries-1","title":"Audio Retrieval with Natural Language Queries: A Benchmark Study","date":"2021-12-17","rows_on_this_dataset":4,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}