{"url":"/dataset/nsynth","name":"NSynth","full_name":"NSynth","description_markdown":"**NSynth** is a dataset of one shot instrumental notes, containing 305,979 musical notes with unique pitch, timbre and envelope. The sounds were collected from 1006 instruments from commercial sample libraries and are annotated based on their source (acoustic, electronic or synthetic), instrument family and sonic qualities. The instrument families used in the annotation are bass, brass, flute, guitar, keyboard, mallet, organ, reed, string, synth lead and vocal. Four second monophonic 16kHz audio snippets were generated (notes) for the instruments.\r\n\r\nSource: [Data Augmentation for Instrument Classification Robust to Audio Effects](https://arxiv.org/abs/1907.08520)\r\nImage Source: [https://magenta.tensorflow.org/nsynth](https://magenta.tensorflow.org/nsynth)","description_withheld":null,"homepage":"https://magenta.tensorflow.org/datasets/nsynth","introduced_date":"2017-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/neural-audio-synthesis-of-musical-notes-with","title":"Neural Audio Synthesis of Musical Notes with WaveNet Autoencoders","first_author":"Jesse Engel","url":null},"license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"modalities":[{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Self-Supervised Learning","url":"/task/self-supervised-learning","datasets_with_task":"/datasets/task/self-supervised-learning"},{"name":"Few-Shot Audio Classification","url":"/task/few-shot-audio-classification","datasets_with_task":"/datasets/task/few-shot-audio-classification"},{"name":"Audio Generation","url":"/task/audio-generation","datasets_with_task":"/datasets/task/audio-generation"},{"name":"Instrument Recognition","url":"/task/instrument-recognition","datasets_with_task":"/datasets/task/instrument-recognition"},{"name":"Music Generation","url":"/task/music-generation","datasets_with_task":"/datasets/task/music-generation"},{"name":"Pitch Classification","url":"/task/pitch-classification","datasets_with_task":"/datasets/task/pitch-classification"}],"languages":[],"variants":["NSynth"],"data_loaders":[{"repo":"https://github.com/activeloopai/Hub","url":"https://docs.activeloop.ai/datasets/nsynth-dataset","frameworks":["tf","pytorch"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/nsynth","frameworks":["tf","jax"]}],"num_papers_in_archive":138,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/few-shot-audio-classification-on-nsynth","task":"Few-Shot Audio Classification","dataset_variant":"NSynth","rows":10,"metrics":["Top-1 Accuracy(5-Way-1-Shot)"],"first_row_in_archive_order":{"model":"Meta-Curvature (CRNN)","paper":"/paper/metaaudio-a-few-shot-audio-classification","metrics":{"Top-1 Accuracy(5-Way-1-Shot)":"96.47 +-0.19"},"code_links":[{"title":"cheggan/metaaudio-a-few-shot-audio-classification-benchmark","url":"https://github.com/cheggan/metaaudio-a-few-shot-audio-classification-benchmark"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/instrument-recognition-on-nsynth","task":"Instrument Recognition","dataset_variant":"NSynth","rows":7,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"M2D-CLAP","paper":"/paper/m2d2-exploring-general-purpose-audio-language","metrics":{"Accuracy":"80.6"},"code_links":[{"title":"nttcslab/m2d","url":"https://github.com/nttcslab/m2d"},{"title":"nttcslab/eval-audio-repr","url":"https://github.com/nttcslab/eval-audio-repr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/m2d2-exploring-general-purpose-audio-language","title":"M2D2: Exploring General-purpose Audio-Language Representations Beyond CLAP","date":"2025-03-28","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/masked-latent-prediction-and-classification","title":"Masked Latent Prediction and Classification for Self-Supervised Audio Representation Learning","date":"2025-02-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mt-slvr-multi-task-self-supervised-learning","title":"MT-SLVR: Multi-Task Self-Supervised Learning for Transformation In(Variant) Representations","date":"2023-05-29","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/efficientleaf-a-faster-learnable-audio","title":"EfficientLEAF: A Faster LEarnable Audio Frontend of Questionable Use","date":"2022-07-12","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/metaaudio-a-few-shot-audio-classification","title":"MetaAudio: A Few-Shot Audio Classification Benchmark","date":"2022-04-05","rows_on_this_dataset":7,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}