{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/the-song-describer-dataset-a-corpus-of-audio","title":"The Song Describer Dataset: a Corpus of Audio Captions for Music-and-Language Evaluation","arxiv_id":"2311.10057","date":"2023-11-16","proceeding":null,"authors":["Ilaria Manco","Benno Weck","Seungheon Doh","Minz Won","Yixiao Zhang","Dmitry Bogdanov","Yusong Wu","Ke Chen","Philip Tovstogan","Emmanouil Benetos","Elio Quinton","György Fazekas","Juhan Nam"],"abstract":"We introduce the Song Describer dataset (SDD), a new crowdsourced corpus of high-quality audio-caption pairs, designed for the evaluation of music-and-language models. The dataset consists of 1.1k human-written natural language descriptions of 706 music recordings, all publicly accessible and released under Creative Common licenses. To showcase the use of our dataset, we benchmark popular models on three key music-and-language tasks (music captioning, text-to-music generation and music-language retrieval). Our experiments highlight the importance of cross-dataset evaluation and offer insights into how researchers can use SDD to gain a broader understanding of model performance.","url_abs":"https://arxiv.org/abs/2311.10057v3","url_pdf":"https://arxiv.org/pdf/2311.10057v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"the-song-describer-dataset-a-corpus-of-audio","repo_url":"https://github.com/mulab-mir/song-describer-dataset","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"music-captioning","task_name":"Music Captioning"},{"task_slug":"music-generation","task_name":"Music Generation"},{"task_slug":"retrieval","task_name":"Retrieval"},{"task_slug":"text-to-audio-retrieval","task_name":"Text to Audio Retrieval"},{"task_slug":"text-to-music-generation","task_name":"Text-to-Music Generation"}],"methods":[],"datasets_introduced":[{"slug":"song-describer-dataset","name":"Song Describer Dataset","full_name":""}],"methods_introduced":[],"results":[],"syntology":{"syntology_url":null,"atlas_url":"https://app.syntology.ai/?focus=2311.10057","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10057"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"deterministic:regex_extraction","url":"https://github.com/ilaria-manco/song-describer","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/mulab-mir/song-describer-dataset","reach":{"status":"ok","spdx":"MIT"}}],"summary":{"ran":3},"by_repo_kind":{"found_in_text":{"samples":3,"ran":3,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"bca18f104d2a57bd","entry":"get_id","repo":"ilaria-manco/song-describer","repo_kind":"found_in_text","path":"annotation_tool/backend/utils.py","file_url":"https://github.com/ilaria-manco/song-describer/blob/HEAD/annotation_tool/backend/utils.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"bca18f104d2a57bd"}},{"code_sha256_prefix":"658cbcdf12d20ff3","entry":"read_file","repo":"ilaria-manco/song-describer","repo_kind":"found_in_text","path":"annotation_tool/backend/utils.py","file_url":"https://github.com/ilaria-manco/song-describer/blob/HEAD/annotation_tool/backend/utils.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"658cbcdf12d20ff3"}},{"code_sha256_prefix":"13b6cda5a36e4c07","entry":"read_lincense_file","repo":"ilaria-manco/song-describer","repo_kind":"found_in_text","path":"annotation_tool/backend/utils.py","file_url":"https://github.com/ilaria-manco/song-describer/blob/HEAD/annotation_tool/backend/utils.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"13b6cda5a36e4c07"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}