{"url":"/sota/audio-classification-on-fsd50k","task":{"name":"Audio Classification","url":"/task/audio-classification","note":null},"dataset":{"name":"FSD50K","url":"/dataset/fsd50k"},"category":"Audio","categories":["Audio"],"category_note":null,"description":"**Audio Classification** is a machine learning task that involves identifying and tagging audio signals into different classes or categories. The goal of audio classification is to enable machines to automatically recognize and distinguish between different types of audio, such as music, speech, and environmental sounds.","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["mAP","Mean AP"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"mAP":"higher","Mean AP":"higher"}},"counts":{"rows":10,"rows_with_code":7,"rows_with_paper_page":10,"rows_dated":10,"rows_using_additional_data":6},"rows":[{"rank_in_archive_order":1,"model":"ONE-PEACE","metrics":{"mAP":"69.7"},"uses_additional_data":true,"paper_date":"2023-05-18","paper":"/paper/one-peace-exploring-one-general","paper_url":"https://arxiv.org/abs/2305.11172v1","paper_title":"ONE-PEACE: Exploring One General Representation Model Toward Unlimited Modalities","code":"https://github.com/modelscope/modelscope","n_code_links":2,"syntology":{"n_ran":2,"n_unverified":5,"n_samples":7,"n_pointer_only_licence":0}},{"rank_in_archive_order":2,"model":"MN","metrics":{"mAP":"65.6"},"uses_additional_data":true,"paper_date":"2023-10-24","paper":"/paper/dynamic-convolutional-neural-networks-as","paper_url":"https://arxiv.org/abs/2310.15648v1","paper_title":"Dynamic Convolutional Neural Networks as Efficient Pre-trained Audio Models","code":"https://github.com/fschmid56/efficientat","n_code_links":1,"syntology":null},{"rank_in_archive_order":3,"model":"PaSST-S","metrics":{"mAP":"65.55"},"uses_additional_data":true,"paper_date":"2021-10-11","paper":"/paper/efficient-training-of-audio-transformers-with","paper_url":"https://arxiv.org/abs/2110.05069v3","paper_title":"Efficient Training of Audio Transformers with Patchout","code":"https://github.com/kkoutini/passt","n_code_links":2,"syntology":{"n_ran":3,"n_unverified":0,"n_samples":3,"n_pointer_only_licence":0}},{"rank_in_archive_order":4,"model":"DyMN-L","metrics":{"mAP":"65.5"},"uses_additional_data":true,"paper_date":"2023-10-24","paper":"/paper/dynamic-convolutional-neural-networks-as","paper_url":"https://arxiv.org/abs/2310.15648v1","paper_title":"Dynamic Convolutional Neural Networks as Efficient Pre-trained Audio Models","code":"https://github.com/fschmid56/efficientat","n_code_links":1,"syntology":null},{"rank_in_archive_order":5,"model":"PaSST-N-S","metrics":{"mAP":"64.2"},"uses_additional_data":true,"paper_date":"2021-10-11","paper":"/paper/efficient-training-of-audio-transformers-with","paper_url":"https://arxiv.org/abs/2110.05069v3","paper_title":"Efficient Training of Audio Transformers with Patchout","code":"https://github.com/kkoutini/passt","n_code_links":2,"syntology":{"n_ran":3,"n_unverified":0,"n_samples":3,"n_pointer_only_licence":0}},{"rank_in_archive_order":6,"model":"PSLA","metrics":{"mAP":"56.71"},"uses_additional_data":true,"paper_date":"2021-02-02","paper":"/paper/psla-improving-audio-event-classification","paper_url":"https://arxiv.org/abs/2102.01243v3","paper_title":"PSLA: Improving Audio Tagging with Pretraining, Sampling, Labeling, and Aggregation","code":"https://github.com/YuanGongND/psla","n_code_links":1,"syntology":null},{"rank_in_archive_order":7,"model":"MATPAC (SSL Model)","metrics":{"mAP":"55.2"},"uses_additional_data":false,"paper_date":"2025-02-17","paper":"/paper/masked-latent-prediction-and-classification","paper_url":"https://arxiv.org/abs/2502.12031v1","paper_title":"Masked Latent Prediction and Classification for Self-Supervised Audio Representation Learning","code":"https://github.com/aurianworld/matpac","n_code_links":1,"syntology":null},{"rank_in_archive_order":8,"model":"Temporal Knowledge Distillation for On-device Audio Classification","metrics":{"mAP":"54.8"},"uses_additional_data":false,"paper_date":"2021-10-27","paper":"/paper/temporal-knowledge-distillation-for-on-device","paper_url":"https://arxiv.org/abs/2110.14131v2","paper_title":"Temporal Knowledge Distillation for On-device Audio Classification","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":9,"model":"Large 6-Layer Transformer with Pooling","metrics":{"mAP":"53.7"},"uses_additional_data":false,"paper_date":"2021-05-01","paper":"/paper/audio-transformers-transformer-architectures","paper_url":"https://arxiv.org/abs/2105.00335v2","paper_title":"Audio Transformers","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":10,"model":"LHGNN","metrics":{"Mean AP":"59"},"uses_additional_data":false,"paper_date":"2025-01-07","paper":"/paper/lhgnn-local-higher-order-graph-neural","paper_url":"https://arxiv.org/abs/2501.03464v2","paper_title":"LHGNN: Local-Higher Order Graph Neural Networks For Audio Classification and Tagging","code":null,"n_code_links":0,"syntology":null}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":3,"rows_with_any_sample_ran":3,"distinct_papers_with_graph_line":2,"distinct_papers_with_any_sample_ran":2,"samples_over_distinct_papers":{"n_ran":5,"n_unverified":5,"n_samples":10,"n_pointer_only_licence":0,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":8,"n_unverified":5,"n_samples":13,"n_pointer_only_licence":0,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}