{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/audio-classification/papers/4","list_of":"/task/audio-classification","task":"Audio Classification","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":4,"rows_per_page":100,"rows":[301,361],"of":361,"counts":{"archive_papers_tagged":361,"with_a_code_link":183,"where_syntology_ran_a_sample":46,"not_listed_spam_title":0,"listed":361,"listed_where_code_ran":46,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":41,"every_run_a_failure_of_syntologys_instrument":5,"listed_with_a_run_with_no_instrument_failure":41,"listed_every_run_a_failure_of_syntologys_instrument":5,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/audio-classification","prev":"/task/audio-classification/papers/3","next":null,"papers":[{"url":null,"slug":"machine-learning-based-covid-19-detection","title":"COVID-19 Detection in Cough, Breath and Speech using Deep Transfer Learning and Bottleneck Features","date":"2021-04-02","arxiv_id":"2104.02477","repositories_listed":0,"syntology":null},{"url":"/paper/self-paced-ensemble-learning-for-speech-and","slug":"self-paced-ensemble-learning-for-speech-and","title":"Self-paced ensemble learning for speech and audio classification","date":"2021-03-22","arxiv_id":"2103.11988","repositories_listed":0,"syntology":null},{"url":"/paper/accurate-and-efficient-time-domain","slug":"accurate-and-efficient-time-domain","title":"Accurate and efficient time-domain classification with adaptive spiking recurrent neural networks","date":"2021-03-12","arxiv_id":"2103.12593","repositories_listed":0,"syntology":null},{"url":"/paper/multi-format-contrastive-learning-of-audio","slug":"multi-format-contrastive-learning-of-audio","title":"Multi-Format Contrastive Learning of Audio Representations","date":"2021-03-11","arxiv_id":"2103.06508","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-weighting-scheme-for-automatic-time","title":"Adaptive Weighting Scheme for Automatic Time-Series Data Augmentation","date":"2021-02-16","arxiv_id":"2102.08310","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-learning-for-few-shot-audio","title":"Semi Supervised Learning For Few-shot Audio Classification By Episodic Triplet Mining","date":"2021-02-16","arxiv_id":"2102.08074","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-convolutional-and-recurrent-networks-for","title":"Deep Convolutional and Recurrent Networks for Polyphonic Instrument Classification from Monophonic Raw Audio Waveforms","date":"2021-02-13","arxiv_id":"2102.06930","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-audio-augmentation-methods-with","title":"Enhancing Audio Augmentation Methods with Consistency Learning","date":"2021-02-09","arxiv_id":"2102.05151","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-universal-learnable-audio-frontend","title":"A Universal Learnable Audio Frontend","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-of-few-shot-audio-classification","title":"A Study of Few-Shot Audio Classification","date":"2020-12-02","arxiv_id":"2012.01573","repositories_listed":0,"syntology":null},{"url":null,"slug":"covid-19-cough-classification-using-machine","title":"COVID-19 Cough Classification using Machine Learning and Global Smartphone Recordings","date":"2020-12-02","arxiv_id":"2012.01926","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-audio-classification-with-factored","title":"Zero-Shot Audio Classification with Factored Linear and Nonlinear Acoustic-Semantic Projections","date":"2020-11-25","arxiv_id":"2011.12657","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-audio-classification-via-semantic","title":"Zero-Shot Audio Classification via Semantic Embeddings","date":"2020-11-24","arxiv_id":"2011.12133","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-self-supervision-from-generalized-1","title":"Multi-modal Self-Supervision from Generalized Data Transformations","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-aedes-aegypti-mosquitoes-through","title":"Detecting Aedes Aegypti Mosquitoes through Audio Classification with Convolutional Neural Networks","date":"2020-08-19","arxiv_id":"2008.09024","repositories_listed":0,"syntology":null},{"url":"/paper/lungrn-nl-an-improved-adventitious-lung-sound","slug":"lungrn-nl-an-improved-adventitious-lung-sound","title":"LungRN+NL: An Improved Adventitious Lung Sound Classification Using Non-Local Block ResNet Neural Network with Mixup Data Augmentation","date":"2020-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-ensemble-of-convolutional-neural-networks","title":"An Ensemble of Convolutional Neural Networks for Audio Classification","date":"2020-07-15","arxiv_id":"2007.07966","repositories_listed":0,"syntology":null},{"url":null,"slug":"private-speech-characterization-with-secure","title":"Private Speech Classification with Secure Multiparty Computation","date":"2020-07-01","arxiv_id":"2007.00253","repositories_listed":0,"syntology":null},{"url":"/paper/a-sequential-self-teaching-approach-for","slug":"a-sequential-self-teaching-approach-for","title":"A Sequential Self Teaching Approach for Improving Generalization in Sound Event Recognition","date":"2020-06-30","arxiv_id":"2007.00144","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-neural-network-for-audio","title":"A Deep Neural Network for Audio Classification with a Classifier Attention Mechanism","date":"2020-06-14","arxiv_id":"2006.09815","repositories_listed":0,"syntology":null},{"url":"/paper/large-scale-audiovisual-learning-of-sounds","slug":"large-scale-audiovisual-learning-of-sounds","title":"Large Scale Audiovisual Learning of Sounds with Weakly Labeled Data","date":"2020-05-29","arxiv_id":"2006.01595","repositories_listed":0,"syntology":null},{"url":null,"slug":"microphone-array-based-surveillance-audio","title":"Microphone Array Based Surveillance Audio Classification","date":"2020-05-22","arxiv_id":"2005.11348","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-for-robustness-certification-of","title":"A FRAMEWORK FOR ROBUSTNESS CERTIFICATION OF SMOOTHED CLASSIFIERS USING F-DIVERGENCES","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-classification-of-large-scale","title":"Automatic Classification of Large-Scale Respiratory Sound Dataset Based on Convolutional Neural Network","date":"2020-01-30","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-context-attention-and-audio","title":"Exploring Context, Attention and Audio Features for Audio Visual Scene-Aware Dialog","date":"2019-12-20","arxiv_id":"1912.10132","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-topics-and-audio-features-with","title":"Leveraging Topics and Audio Features with Multimodal Attention for Audio Visual Scene-Aware Dialog","date":"2019-12-20","arxiv_id":"1912.10131","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-augmentation-approaches-for-improving","title":"Data augmentation approaches for improving animal audio classification","date":"2019-12-16","arxiv_id":"1912.07756","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-discriminative-and-robust-time","title":"Environmental Sound Classification with Parallel Temporal-spectral Attention","date":"2019-12-14","arxiv_id":"1912.06808","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-end-to-end-audio-classification-system","title":"An End-to-End Audio Classification System based on Raw Waveforms and Mix-Training Strategy","date":"2019-11-21","arxiv_id":"1911.09349","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-supervised-learning-for-better","title":"Cross-modal supervised learning for better acoustic representations","date":"2019-11-15","arxiv_id":"1911.07917","repositories_listed":0,"syntology":null},{"url":null,"slug":"segment-relevance-estimation-for-audio","title":"Segment Relevance Estimation for Audio Analysis and Weakly-Labelled Classification","date":"2019-11-12","arxiv_id":"1911.04666","repositories_listed":0,"syntology":null},{"url":null,"slug":"label-efficient-audio-classification-through","title":"Label-efficient audio classification through multitask learning and self-supervision","date":"2019-10-19","arxiv_id":"1910.12587","repositories_listed":0,"syntology":null},{"url":"/paper/the-heidelberg-spiking-datasets-for-the","slug":"the-heidelberg-spiking-datasets-for-the","title":"The Heidelberg spiking datasets for the systematic evaluation of spiking neural networks","date":"2019-10-16","arxiv_id":"1910.07407","repositories_listed":0,"syntology":null},{"url":null,"slug":"certifying-neural-network-audio-classifiers","title":"Certifying Neural Network Audio Classifiers","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"defensive-tensorization-randomized-tensor","title":"Defensive Tensorization: Randomized Tensor Parametrization for Robust Neural Networks","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"spectrobank-a-filter-bank-convolutional-layer","title":"SpectroBank: A filter-bank convolutional layer for CNN-based audio applications","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-for-earth-rainforest-conservation-by","title":"AI for Earth: Rainforest Conservation by Acoustic Surveillance","date":"2019-08-20","arxiv_id":"1908.07517","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-neural-baselines-for-computational","title":"Deep Neural Baselines for Computational Paralinguistics","date":"2019-07-05","arxiv_id":"1907.02864","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-performance-of-residual-block-design","title":"On the performance of residual block design alternatives in convolutional neural networks for end-to-end audio classification","date":"2019-06-26","arxiv_id":"1906.10891","repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneously-learning-architectures-and","title":"Simultaneously Learning Architectures and Features of Deep Neural Networks","date":"2019-06-11","arxiv_id":"1906.04505","repositories_listed":0,"syntology":null},{"url":null,"slug":"dcase-2019-cnn-depth-analysis-with-different","title":"CNN depth analysis with different channel inputs for Acoustic Scene Classification","date":"2019-06-10","arxiv_id":"1906.04591","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-audio-classification-based-on-class","title":"Zero-Shot Audio Classification Based on Class Label Embeddings","date":"2019-05-06","arxiv_id":"1905.01926","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-robust-approach-for-securing-audio","title":"A Robust Approach for Securing Audio Classification Against Adversarial Attacks","date":"2019-04-24","arxiv_id":"1904.10990","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-classification-of-bit-representation","title":"Audio Classification of Bit-Representation Waveform","date":"2019-04-08","arxiv_id":"1904.04364","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-performance-and-inference-on-audio","title":"Improving performance and inference on audio classification tasks using capsule networks","date":"2019-02-13","arxiv_id":"1902.05069","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-attention-and-audio-feature","title":"Context, Attention and Audio Feature Explorations for Audio Visual Scene-Aware Dialog","date":"2018-12-20","arxiv_id":"1812.08407","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-domain-deep-feature-combination-for","title":"Cross-domain Deep Feature Combination for Bird Species Classification with Audio-visual Data","date":"2018-11-26","arxiv_id":"1811.10199","repositories_listed":0,"syntology":null},{"url":null,"slug":"aclnet-efficient-end-to-end-audio","title":"AclNet: efficient end-to-end audio classification CNN","date":"2018-11-16","arxiv_id":"1811.06669","repositories_listed":0,"syntology":null},{"url":"/paper/cooperative-learning-of-audio-and-video","slug":"cooperative-learning-of-audio-and-video","title":"Cooperative Learning of Audio and Video Models from Self-Supervised Synchronization","date":"2018-06-30","arxiv_id":"1807.00230","repositories_listed":0,"syntology":null},{"url":"/paper/wsnet-learning-compact-and-efficient-networks","slug":"wsnet-learning-compact-and-efficient-networks","title":"WSNet: Learning Compact and Efficient Networks with Weight Sampling","date":"2018-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"representations-of-sound-in-deep-learning-of","title":"Representations of Sound in Deep Learning of Audio Features from Music","date":"2017-12-08","arxiv_id":"1712.02898","repositories_listed":0,"syntology":null},{"url":null,"slug":"raw-waveform-based-audio-classification-using","title":"Raw Waveform-based Audio Classification Using Sample-level CNN Architectures","date":"2017-12-04","arxiv_id":"1712.00866","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-sparse-adversarial-dictionaries-for","title":"Learning Sparse Adversarial Dictionaries For Multi-Class Audio Classification","date":"2017-12-02","arxiv_id":"1712.00640","repositories_listed":0,"syntology":null},{"url":null,"slug":"fearnet-brain-inspired-model-for-incremental","title":"FearNet: Brain-Inspired Model for Incremental Learning","date":"2017-11-28","arxiv_id":"1711.10563","repositories_listed":0,"syntology":null},{"url":null,"slug":"wsnet-compact-and-efficient-networks-through","title":"WSNet: Compact and Efficient Networks Through Weight Sampling","date":"2017-11-28","arxiv_id":"1711.10067","repositories_listed":0,"syntology":null},{"url":"/paper/unsupervised-learning-of-semantic-audio","slug":"unsupervised-learning-of-semantic-audio","title":"Unsupervised Learning of Semantic Audio Representations","date":"2017-11-06","arxiv_id":"1711.02209","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparison-of-time-frequency-representations","title":"Comparison of Time-Frequency Representations for Environmental Sound Classification using Convolutional Neural Networks","date":"2017-06-22","arxiv_id":"1706.07156","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-structure-discovery-for-semantic","title":"Unsupervised Structure Discovery for Semantic Analysis of Audio","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-labeling-and-transcription-with-the","title":"Fast Labeling and Transcription with the Speechalyzer Toolkit","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-filtering","title":"Sparse Filtering","date":"2011-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-feature-learning-for-audio","title":"Unsupervised feature learning for audio classification using convolutional deep belief networks","date":"2009-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"ee342fd3c142b8e02faff3fcc0e425ee3139179be19f2350c6f0b034a6842b84","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}