{"url":"/task/few-shot-audio-classification","name":"Few-Shot Audio Classification","slug":"few-shot-audio-classification","description_markdown":"Few-shot classification for audio signals. Presents a unique challenge compared to other few-shot domains as we deal with temporal dependencies as well.\r\n\r\nLike other few-shot problems, few-shot audio classification can be tackled in a  variety of ways, from using supervised meta-learning on the same primary dataset, to pre-training on an external dataset and utilising linear readout. For this reason, results in each dataset leaderboard should be correctly tagged e.g. with \"Within Dataset Meta-Learning\" etc","categories":[{"name":"Audio","url":"/area/audio"},{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Methodology","url":"/area/methodology"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":8,"papers_with_code":5,"benchmarks":10,"benchmark_tables_in_archive":10,"benchmark_tables_shown":10,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":9,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/few-shot-audio-classification-on","slug":"few-shot-audio-classification-on","dataset":"FSDKaggle2018","dataset_url":"/dataset/fsdkaggle2018","rows_in_archive":10,"metrics":["Top-1 Accuracy(5-Way-1-Shot)"],"first_row_in_archive_order":{"model":"MAML (CRNN)","paper_title":"MetaAudio: A Few-Shot Audio Classification Benchmark","paper_url":"/paper/metaaudio-a-few-shot-audio-classification","paper_date":"2022-04-05","arxiv_id":"2204.02121","code_links":[{"title":"cheggan/metaaudio-a-few-shot-audio-classification-benchmark","url":"https://github.com/cheggan/metaaudio-a-few-shot-audio-classification-benchmark"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}}},{"leaderboard":"/sota/few-shot-audio-classification-on-birdclef","slug":"few-shot-audio-classification-on-birdclef","dataset":"BirdClef 2020  (Pruned)","dataset_url":"/dataset/birdclef-2020-pruned","rows_in_archive":10,"metrics":["Top-1 Accuracy(5-Way-1-Shot)"],"first_row_in_archive_order":{"model":"Meta-Curvature (CRNN)","paper_title":"MetaAudio: A Few-Shot Audio Classification Benchmark","paper_url":"/paper/metaaudio-a-few-shot-audio-classification","paper_date":"2022-04-05","arxiv_id":"2204.02121","code_links":[{"title":"cheggan/metaaudio-a-few-shot-audio-classification-benchmark","url":"https://github.com/cheggan/metaaudio-a-few-shot-audio-classification-benchmark"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}}},{"leaderboard":"/sota/few-shot-audio-classification-on-esc-50","slug":"few-shot-audio-classification-on-esc-50","dataset":"ESC-50","dataset_url":"/dataset/esc-50","rows_in_archive":10,"metrics":["Top-1 Accuracy(5-Way-1-Shot)"],"first_row_in_archive_order":{"model":"Meta-Curvature (CRNN)","paper_title":"MetaAudio: A Few-Shot Audio Classification Benchmark","paper_url":"/paper/metaaudio-a-few-shot-audio-classification","paper_date":"2022-04-05","arxiv_id":"2204.02121","code_links":[{"title":"cheggan/metaaudio-a-few-shot-audio-classification-benchmark","url":"https://github.com/cheggan/metaaudio-a-few-shot-audio-classification-benchmark"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}}},{"leaderboard":"/sota/few-shot-audio-classification-on-nsynth","slug":"few-shot-audio-classification-on-nsynth","dataset":"NSynth","dataset_url":"/dataset/nsynth","rows_in_archive":10,"metrics":["Top-1 Accuracy(5-Way-1-Shot)"],"first_row_in_archive_order":{"model":"Meta-Curvature (CRNN)","paper_title":"MetaAudio: A Few-Shot Audio Classification Benchmark","paper_url":"/paper/metaaudio-a-few-shot-audio-classification","paper_date":"2022-04-05","arxiv_id":"2204.02121","code_links":[{"title":"cheggan/metaaudio-a-few-shot-audio-classification-benchmark","url":"https://github.com/cheggan/metaaudio-a-few-shot-audio-classification-benchmark"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}}},{"leaderboard":"/sota/few-shot-audio-classification-on-voxceleb1","slug":"few-shot-audio-classification-on-voxceleb1","dataset":"VoxCeleb1","dataset_url":"/dataset/voxceleb1","rows_in_archive":10,"metrics":["Top-1 Accuracy(5-Way-1-Shot)"],"first_row_in_archive_order":{"model":"Meta-Curvature (CRNN)","paper_title":"MetaAudio: A Few-Shot Audio Classification Benchmark","paper_url":"/paper/metaaudio-a-few-shot-audio-classification","paper_date":"2022-04-05","arxiv_id":"2204.02121","code_links":[{"title":"cheggan/metaaudio-a-few-shot-audio-classification-benchmark","url":"https://github.com/cheggan/metaaudio-a-few-shot-audio-classification-benchmark"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}}},{"leaderboard":"/sota/few-shot-audio-classification-on-watkins","slug":"few-shot-audio-classification-on-watkins","dataset":"Watkins Marine Mammal Sounds","dataset_url":"/dataset/watkins-marine-mammal-sounds","rows_in_archive":5,"metrics":["Top-1 Accuracy(5-Way-1-Shot)"],"first_row_in_archive_order":{"model":"MT-SLVR (SimCLR + MLAP) w/ Parallel Adapters (FSD50K, RN18)","paper_title":"MT-SLVR: Multi-Task Self-Supervised Learning for Transformation In(Variant) Representations","paper_url":"/paper/mt-slvr-multi-task-self-supervised-learning","paper_date":"2023-05-29","arxiv_id":"2305.17191","code_links":[{"title":"cheggan/mt-slvr","url":"https://github.com/cheggan/mt-slvr"}],"syntology":null}},{"leaderboard":"/sota/few-shot-audio-classification-on-common-voice","slug":"few-shot-audio-classification-on-common-voice","dataset":"Common Voice","dataset_url":"/dataset/common-voice","rows_in_archive":3,"metrics":["Top-1 Accuracy(5-Way-1-Shot)"],"first_row_in_archive_order":{"model":"MT-SLVR (SimCLR + MLAP) w/ Parallel Adapters (FSD50K, RN18)","paper_title":"MT-SLVR: Multi-Task Self-Supervised Learning for Transformation In(Variant) Representations","paper_url":"/paper/mt-slvr-multi-task-self-supervised-learning","paper_date":"2023-05-29","arxiv_id":"2305.17191","code_links":[{"title":"cheggan/mt-slvr","url":"https://github.com/cheggan/mt-slvr"}],"syntology":null}},{"leaderboard":"/sota/few-shot-audio-classification-on-crema-d","slug":"few-shot-audio-classification-on-crema-d","dataset":"CREMA-D","dataset_url":"/dataset/crema-d","rows_in_archive":3,"metrics":["Top-1 Accuracy(5-Way-1-Shot)"],"first_row_in_archive_order":{"model":"MT-SLVR (SimCLR + MLAP) w/ Parallel Adapters (FSD50K, RN18)","paper_title":"MT-SLVR: Multi-Task Self-Supervised Learning for Transformation In(Variant) Representations","paper_url":"/paper/mt-slvr-multi-task-self-supervised-learning","paper_date":"2023-05-29","arxiv_id":"2305.17191","code_links":[{"title":"cheggan/mt-slvr","url":"https://github.com/cheggan/mt-slvr"}],"syntology":null}},{"leaderboard":"/sota/few-shot-audio-classification-on-speech","slug":"few-shot-audio-classification-on-speech","dataset":"Speech Command v2","dataset_url":null,"rows_in_archive":3,"metrics":["Top-1 Accuracy(5-Way-1-Shot)"],"first_row_in_archive_order":{"model":"SimCLR (FSD50K, RN18)","paper_title":"MT-SLVR: Multi-Task Self-Supervised Learning for Transformation In(Variant) Representations","paper_url":"/paper/mt-slvr-multi-task-self-supervised-learning","paper_date":"2023-05-29","arxiv_id":"2305.17191","code_links":[{"title":"cheggan/mt-slvr","url":"https://github.com/cheggan/mt-slvr"}],"syntology":null}},{"leaderboard":"/sota/few-shot-audio-classification-on-speech-1","slug":"few-shot-audio-classification-on-speech-1","dataset":"Speech Accent Archive","dataset_url":"/dataset/speech-accent-archive","rows_in_archive":3,"metrics":["Top-1 Accuracy(5-Way-1-Shot)"],"first_row_in_archive_order":{"model":"MT-SLVR (SimCLR + MLAP) w/ Parallel Adapters (FSD50K, RN18)","paper_title":"MT-SLVR: Multi-Task Self-Supervised Learning for Transformation In(Variant) Representations","paper_url":"/paper/mt-slvr-multi-task-self-supervised-learning","paper_date":"2023-05-29","arxiv_id":"2305.17191","code_links":[{"title":"cheggan/mt-slvr","url":"https://github.com/cheggan/mt-slvr"}],"syntology":null}}],"datasets":[{"url":"/dataset/voxceleb1","name":"VoxCeleb1","full_name":"VoxCeleb1","num_papers_in_archive":680},{"url":"/dataset/common-voice","name":"Common Voice","full_name":"Common Voice","num_papers_in_archive":449},{"url":"/dataset/esc-50","name":"ESC-50","full_name":"ESC-50","num_papers_in_archive":387},{"url":"/dataset/nsynth","name":"NSynth","full_name":"NSynth","num_papers_in_archive":138},{"url":"/dataset/crema-d","name":"CREMA-D","full_name":"CREMA-D","num_papers_in_archive":28},{"url":"/dataset/fsdkaggle2018","name":"FSDKaggle2018","full_name":"FSDKaggle2018","num_papers_in_archive":12},{"url":"/dataset/birdclef-2020-pruned","name":"BirdClef 2020  (Pruned)","full_name":"","num_papers_in_archive":3},{"url":"/dataset/speech-accent-archive","name":"Speech Accent Archive","full_name":"The Speech Accent Archive","num_papers_in_archive":2},{"url":"/dataset/watkins-marine-mammal-sounds","name":"Watkins Marine Mammal Sounds","full_name":"Watkins Marine Mammal Sound Database","num_papers_in_archive":2}],"subtasks":[],"parent_tasks":[{"url":"/task/few-shot-learning","name":"Few-Shot Learning"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":5,"of":5,"tagged_in_all":8,"items":[{"url":"/paper/acoustic-prompt-tuning-empowering-large","title":"Acoustic Prompt Tuning: Empowering Large Language Models with Audition Capabilities","date":"2023-11-30","arxiv_id":"2312.00249","repositories_listed":2,"syntology":null},{"url":"/paper/episodic-fine-tuning-prototypical-networks","title":"Episodic fine-tuning prototypical networks for optimization-based few-shot learning: Application to audio classification","date":"2024-10-04","arxiv_id":"2410.05302","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-transferability-of-large-scale-self","title":"On the Transferability of Large-Scale Self-Supervision to Few-Shot Audio Classification","date":"2024-02-02","arxiv_id":"2402.01274","repositories_listed":1,"syntology":null},{"url":"/paper/mt-slvr-multi-task-self-supervised-learning","title":"MT-SLVR: Multi-Task Self-Supervised Learning for Transformation In(Variant) Representations","date":"2023-05-29","arxiv_id":"2305.17191","repositories_listed":1,"syntology":null},{"url":"/paper/metaaudio-a-few-shot-audio-classification","title":"MetaAudio: A Few-Shot Audio Classification Benchmark","date":"2022-04-05","arxiv_id":"2204.02121","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}}],"syntology_records":1,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}