{"url":"/dataset/musicbench","name":"MusicBench","full_name":null,"description_markdown":"The MusicBench dataset is a music audio-text pair dataset that was designed for text-to-music generation purpose and released along with Mustango text-to-music model. MusicBench is based on the MusicCaps dataset, which it expands from 5,521 samples to 52,768 training and 400 test samples!\r\n\r\nDataset Details\r\nMusicBench expands MusicCaps by:\r\n\r\nIncluding music features of chords, beats, tempo, and key that are extracted from the audio.\r\nDescribing these music features using text templates and thus enhancing the original text prompts.\r\nExpanding the number of audio samples by performing musically meaningful augmentations: semitone pitch shifts, tempo changes, and volume changes.\r\n\r\nTrain set size = 52,768 samples Test set size = 400\r\n\r\nThis dataset also includes FMACaps, which was used as a second test set.","description_withheld":null,"homepage":"https://huggingface.co/datasets/amaai-lab/MusicBench","introduced_date":"2023-11-14","introduced_date_note":null,"introduced_by":{"paper":"/paper/mustango-toward-controllable-text-to-music","title":"Mustango: Toward Controllable Text-to-Music Generation","first_author":"Jan Melechovsky","url":null},"license":{"name":"CC","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Audio","url":"/datasets/modality/audio"},{"name":"Music","url":"/datasets/modality/music"}],"tasks":[{"name":"Text-to-Music Generation","url":"/task/text-to-music-generation","datasets_with_task":"/datasets/task/text-to-music-generation"},{"name":"Music Generation","url":"/task/music-generation","datasets_with_task":"/datasets/task/music-generation"}],"languages":[],"variants":["MusicBench"],"data_loaders":[],"num_papers_in_archive":6,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/text-to-music-generation-on-musicbench","task":"Text-to-Music Generation","dataset_variant":"MusicBench","rows":1,"metrics":["FAD"],"first_row_in_archive_order":{"model":"Mustango (non-pretrained)","paper":"/paper/mustango-toward-controllable-text-to-music","metrics":{"FAD":"2.09"},"code_links":[{"title":"amaai-lab/mustango","url":"https://github.com/amaai-lab/mustango"},{"title":"atharva20038/music4all","url":"https://github.com/atharva20038/music4all"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/2/musicgen_melody"},{"title":"MindSpore-scientific/code-10","url":"https://github.com/MindSpore-scientific/code-10/tree/main/text-to-music"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/mustango-toward-controllable-text-to-music","title":"Mustango: Toward Controllable Text-to-Music Generation","date":"2023-11-14","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":7,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":9,"samples_ran":7,"samples_unverified":2,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}