{"url":"/dataset/msd","name":"MSD","full_name":"Million Song Dataset","description_markdown":"The Million Song Dataset is a freely-available collection of audio features and metadata for a million contemporary popular music tracks.\r\n\r\nThe core of the dataset is the feature analysis and metadata for one million songs, provided by The Echo Nest. The dataset does not include any audio, only the derived features. Note, however, that sample audio can be fetched from services like 7digital, using [code]( https://github.com/tbertinmahieux/MSongsDB/tree/master/Tasks_Demos/Preview7digital) provided by the authors.\r\n\r\n\r\nSource: [http://millionsongdataset.com/](http://millionsongdataset.com/)\r\nPaper: [The Million Song Dataset](https://doi.org/10.7916/D8NZ8J07)","description_withheld":null,"homepage":"http://millionsongdataset.com/","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Recommendation Systems","url":"/task/recommendation-systems","datasets_with_task":"/datasets/task/recommendation-systems"},{"name":"Music Auto-Tagging","url":"/task/music-auto-tagging","datasets_with_task":"/datasets/task/music-auto-tagging"}],"languages":[],"variants":["Million Song Dataset","MSD"],"data_loaders":[],"num_papers_in_archive":11,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/collaborative-filtering-on-million-song","task":"Recommendation Systems","dataset_variant":"Million Song Dataset","rows":7,"metrics":["nDCG@100","Recall@50","Recall@20","Recall@100"],"first_row_in_archive_order":{"model":"EASE","paper":"/paper/190503375","metrics":{"Recall@20":"0.333","Recall@50":"0.428","nDCG@100":"0.389"},"code_links":[{"title":"PreferredAI/cornac","url":"https://github.com/PreferredAI/cornac"},{"title":"AmazingDD/daisyRec","url":"https://github.com/AmazingDD/daisyRec"},{"title":"Darel13712/ease_rec","url":"https://github.com/Darel13712/ease_rec"},{"title":"recsys-benchmark/daisyrec-v2.0","url":"https://github.com/recsys-benchmark/daisyrec-v2.0"},{"title":"glami/sansa","url":"https://github.com/glami/sansa"},{"title":"franckjay/TorchEASE","url":"https://github.com/franckjay/TorchEASE"},{"title":"jvbalen/autoencoders_cf","url":"https://github.com/jvbalen/autoencoders_cf"},{"title":"AhmadRK94/NeuEASE","url":"https://github.com/AhmadRK94/NeuEASE"},{"title":"MindSpore-scientific/code-10","url":"https://github.com/MindSpore-scientific/code-10/tree/main/shallow-rnns"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/music-auto-tagging-on-million-song-dataset","task":"Music Auto-Tagging","dataset_variant":"Million Song Dataset","rows":2,"metrics":["PR-AUC","ROC-AUC"],"first_row_in_archive_order":{"model":"CLMR","paper":"/paper/contrastive-learning-of-musical","metrics":{"PR-AUC":"25.0"},"code_links":[{"title":"spijkervet/CLMR","url":"https://github.com/spijkervet/CLMR"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/scalable-approximate-nonsymmetric-autoencoder","title":"Scalable Approximate NonSymmetric Autoencoder for Collaborative Filtering","date":"2023-09-14","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/contrastive-learning-of-musical","title":"Contrastive Learning of Musical Representations","date":"2021-03-17","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/recvae-a-new-variational-autoencoder-for-top","title":"RecVAE: a New Variational Autoencoder for Top-N Recommendations with Implicit Feedback","date":"2019-12-24","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/towards-amortized-ranking-critical-training","title":"Towards Amortized Ranking-Critical Training for Collaborative Filtering","date":"2019-06-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/190503375","title":"Embarrassingly Shallow Autoencoders for Sparse Data","date":"2019-05-08","rows_on_this_dataset":1,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/variational-autoencoders-for-collaborative","title":"Variational Autoencoders for Collaborative Filtering","date":"2018-02-16","rows_on_this_dataset":2,"code_links":18,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":21,"samples_ran":2,"samples_unverified":19,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/collaborative-metric-learning","title":"Collaborative Metric Learning","date":"2017-04-01","rows_on_this_dataset":1,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":31,"samples_ran":6,"samples_unverified":25,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}