{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/audio-classification/papers/2","list_of":"/task/audio-classification","task":"Audio Classification","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":4,"rows_per_page":100,"rows":[101,200],"of":361,"counts":{"archive_papers_tagged":361,"with_a_code_link":183,"where_syntology_ran_a_sample":46,"not_listed_spam_title":0,"listed":361,"listed_where_code_ran":46,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":41,"every_run_a_failure_of_syntologys_instrument":5,"listed_with_a_run_with_no_instrument_failure":41,"listed_every_run_a_failure_of_syntologys_instrument":5,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/audio-classification","prev":"/task/audio-classification","next":"/task/audio-classification/papers/3","papers":[{"url":"/paper/audio-tagging-on-an-embedded-hardware","slug":"audio-tagging-on-an-embedded-hardware","title":"Audio Tagging on an Embedded Hardware Platform","date":"2023-06-15","arxiv_id":"2306.09106","repositories_listed":1,"syntology":null},{"url":"/paper/few-shot-class-incremental-audio-2","slug":"few-shot-class-incremental-audio-2","title":"Few-shot Class-incremental Audio Classification Using Stochastic Classifier","date":"2023-06-03","arxiv_id":"2306.02053","repositories_listed":1,"syntology":null},{"url":"/paper/few-shot-class-incremental-audio-1","slug":"few-shot-class-incremental-audio-1","title":"Few-shot Class-incremental Audio Classification Using Dynamically Expanded Classifier with Self-attention Modified Prototypes","date":"2023-05-31","arxiv_id":"2305.19539","repositories_listed":1,"syntology":null},{"url":"/paper/patch-mix-contrastive-learning-with-audio","slug":"patch-mix-contrastive-learning-with-audio","title":"Patch-Mix Contrastive Learning with Audio Spectrogram Transformer on Respiratory Sound Classification","date":"2023-05-23","arxiv_id":"2305.14032","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/patch-mix-contrastive-learning-with-audio#ran","syntology_url":"https://syntology.ai/paper/2305.14032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14032"}},"official":{"repos":["raymin0223/patch-mix_contrastive_learning"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/device-robust-acoustic-scene-classification-1","slug":"device-robust-acoustic-scene-classification-1","title":"Device-Robust Acoustic Scene Classification via Impulse Response Augmentation","date":"2023-05-12","arxiv_id":"2305.07499","repositories_listed":1,"syntology":null},{"url":"/paper/unfused-unsupervised-finetuning-using-self","slug":"unfused-unsupervised-finetuning-using-self","title":"UNFUSED: UNsupervised Finetuning Using SElf supervised Distillation","date":"2023-03-10","arxiv_id":"2303.05668","repositories_listed":1,"syntology":null},{"url":"/paper/face-fast-accurate-and-context-aware-audio","slug":"face-fast-accurate-and-context-aware-audio","title":"Face: Fast, Accurate and Context-Aware Audio Annotation and Classification","date":"2023-03-07","arxiv_id":"2303.03666","repositories_listed":1,"syntology":null},{"url":"/paper/fosi-hybrid-first-and-second-order","slug":"fosi-hybrid-first-and-second-order","title":"FOSI: Hybrid First and Second Order Optimization","date":"2023-02-16","arxiv_id":"2302.08484","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/fosi-hybrid-first-and-second-order#ran","syntology_url":"https://syntology.ai/paper/2302.08484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.08484"}},"official":{"repos":["hsivan/fosi"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/multimodality-helps-unimodality-cross-modal","slug":"multimodality-helps-unimodality-cross-modal","title":"Multimodality Helps Unimodality: Cross-Modal Few-Shot Learning with Multimodal Models","date":"2023-01-16","arxiv_id":"2301.06267","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/multimodality-helps-unimodality-cross-modal#ran","syntology_url":"https://syntology.ai/paper/2301.06267","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.06267"}},"official":{"repos":["linzhiqiu/cross_modal_adaptation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-spatio-temporal-deep-learning-approach-for","slug":"a-spatio-temporal-deep-learning-approach-for","title":"A Spatio-temporal Deep Learning Approach for Underwater Acoustic Signals Classification","date":"2022-11-24","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/effective-audio-classification-network-based","slug":"effective-audio-classification-network-based","title":"Effective Audio Classification Network Based on Paired Inverse Pyramid Structure and Dense MLP Block","date":"2022-11-05","arxiv_id":"2211.02940","repositories_listed":1,"syntology":null},{"url":"/paper/mast-multiscale-audio-spectrogram","slug":"mast-multiscale-audio-spectrogram","title":"MAST: Multiscale Audio Spectrogram Transformers","date":"2022-11-02","arxiv_id":"2211.01515","repositories_listed":1,"syntology":null},{"url":"/paper/slicer-learning-universal-audio","slug":"slicer-learning-universal-audio","title":"SLICER: Learning universal audio representations using low-resource self-supervised pre-training","date":"2022-11-02","arxiv_id":"2211.01519","repositories_listed":1,"syntology":null},{"url":"/paper/supervised-contrastive-learning-for-3","slug":"supervised-contrastive-learning-for-3","title":"Pretraining Respiratory Sound Representations using Metadata and Contrastive Learning","date":"2022-10-27","arxiv_id":"2210.16192","repositories_listed":1,"syntology":null},{"url":"/paper/masked-modeling-duo-learning-representations","slug":"masked-modeling-duo-learning-representations","title":"Masked Modeling Duo: Learning Representations by Encouraging Both Networks to Model the Input","date":"2022-10-26","arxiv_id":"2210.14648","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/masked-modeling-duo-learning-representations#ran","syntology_url":"https://syntology.ai/paper/2210.14648","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.14648"}},"official":{"repos":["nttcslab/m2d"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/play-it-back-iterative-attention-for-audio","slug":"play-it-back-iterative-attention-for-audio","title":"Play It Back: Iterative Attention for Audio Recognition","date":"2022-10-20","arxiv_id":"2210.11328","repositories_listed":1,"syntology":null},{"url":"/paper/stsc-snn-spatio-temporal-synaptic-connection","slug":"stsc-snn-spatio-temporal-synaptic-connection","title":"STSC-SNN: Spatio-Temporal Synaptic Connection with Temporal Convolution and Attention for Spiking Neural Networks","date":"2022-10-11","arxiv_id":"2210.05241","repositories_listed":1,"syntology":null},{"url":"/paper/learning-the-spectrogram-temporal-resolution","slug":"learning-the-spectrogram-temporal-resolution","title":"Learning Temporal Resolution in Spectrogram for Audio Classification","date":"2022-10-04","arxiv_id":"2210.01719","repositories_listed":1,"syntology":null},{"url":"/paper/simple-pooling-front-ends-for-efficient-audio","slug":"simple-pooling-front-ends-for-efficient-audio","title":"Simple Pooling Front-ends For Efficient Audio Classification","date":"2022-10-03","arxiv_id":"2210.00943","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-audio-visual-masked-autoencoder","slug":"contrastive-audio-visual-masked-autoencoder","title":"Contrastive Audio-Visual Masked Autoencoder","date":"2022-10-02","arxiv_id":"2210.07839","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/contrastive-audio-visual-masked-autoencoder#ran","syntology_url":"https://syntology.ai/paper/2210.07839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07839"}},"official":{"repos":["yuangongnd/cav-mae"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/sub-mw-neuromorphic-snn-audio-processing","slug":"sub-mw-neuromorphic-snn-audio-processing","title":"Sub-mW Neuromorphic SNN audio processing applications with Rockpool and Xylo","date":"2022-08-27","arxiv_id":"2208.12991","repositories_listed":1,"syntology":null},{"url":"/paper/a-study-on-broadcast-networks-for-music-genre","slug":"a-study-on-broadcast-networks-for-music-genre","title":"A Study on Broadcast Networks for Music Genre Classification","date":"2022-08-25","arxiv_id":"2208.12086","repositories_listed":1,"syntology":null},{"url":"/paper/a-surrogate-gradient-spiking-baseline-for","slug":"a-surrogate-gradient-spiking-baseline-for","title":"A surrogate gradient spiking baseline for speech command recognition","date":"2022-08-22","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deep-feature-learning-for-medical-acoustics","slug":"deep-feature-learning-for-medical-acoustics","title":"Deep Feature Learning for Medical Acoustics","date":"2022-08-05","arxiv_id":"2208.03084","repositories_listed":1,"syntology":null},{"url":"/paper/uavm-a-unified-model-for-audio-visual","slug":"uavm-a-unified-model-for-audio-visual","title":"UAVM: Towards Unifying Audio and Visual Models","date":"2022-07-29","arxiv_id":"2208.00061","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/uavm-a-unified-model-for-audio-visual#ran","syntology_url":"https://syntology.ai/paper/2208.00061","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.00061"}},"official":{"repos":["YuanGongND/uavm"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/efficientleaf-a-faster-learnable-audio","slug":"efficientleaf-a-faster-learnable-audio","title":"EfficientLEAF: A Faster LEarnable Audio Frontend of Questionable Use","date":"2022-07-12","arxiv_id":"2207.05508","repositories_listed":1,"syntology":null},{"url":"/paper/uncertainty-calibration-for-deep-audio","slug":"uncertainty-calibration-for-deep-audio","title":"Uncertainty Calibration for Deep Audio Classifiers","date":"2022-06-27","arxiv_id":"2206.13071","repositories_listed":1,"syntology":null},{"url":"/paper/fluctuation-driven-initialization-for-spiking","slug":"fluctuation-driven-initialization-for-spiking","title":"Fluctuation-driven initialization for spiking neural network training","date":"2022-06-21","arxiv_id":"2206.10226","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/fluctuation-driven-initialization-for-spiking#ran","syntology_url":"https://syntology.ai/paper/2206.10226","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.10226"}},"official":{"repos":["fmi-basel/stork"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/accelerating-spiking-neural-network-training","slug":"accelerating-spiking-neural-network-training","title":"Robust and accelerated single-spike spiking neural network training with applicability to challenging temporal tasks","date":"2022-05-30","arxiv_id":"2205.15286","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/accelerating-spiking-neural-network-training#ran","syntology_url":"https://syntology.ai/paper/2205.15286","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.15286"}},"official":{"repos":["webstorms/fastsnn"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lerac-learning-rate-curriculum","slug":"lerac-learning-rate-curriculum","title":"Learning Rate Curriculum","date":"2022-05-18","arxiv_id":"2205.09180","repositories_listed":1,"syntology":null},{"url":"/paper/sound2synth-interpreting-sound-via-fm","slug":"sound2synth-interpreting-sound-via-fm","title":"Sound2Synth: Interpreting Sound via FM Synthesizer Parameters Estimation","date":"2022-05-06","arxiv_id":"2205.03043","repositories_listed":1,"syntology":null},{"url":"/paper/vocalsound-a-dataset-for-improving-human","slug":"vocalsound-a-dataset-for-improving-human","title":"Vocalsound: A Dataset for Improving Human Vocal Sounds Recognition","date":"2022-05-06","arxiv_id":"2205.03433","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-audio-strikes-back-boosting","slug":"end-to-end-audio-strikes-back-boosting","title":"End-to-End Audio Strikes Back: Boosting Augmentations Towards An Efficient Audio Classification Network","date":"2022-04-25","arxiv_id":"2204.11479","repositories_listed":1,"syntology":null},{"url":"/paper/map-snn-mapping-spike-activities-with","slug":"map-snn-mapping-spike-activities-with","title":"MAP-SNN: Mapping Spike Activities with Multiplicity, Adaptability, and Plasticity into Bio-Plausible Spiking Neural Networks","date":"2022-04-21","arxiv_id":"2204.09893","repositories_listed":1,"syntology":null},{"url":"/paper/metaaudio-a-few-shot-audio-classification","slug":"metaaudio-a-few-shot-audio-classification","title":"MetaAudio: A Few-Shot Audio Classification Benchmark","date":"2022-04-05","arxiv_id":"2204.02121","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/metaaudio-a-few-shot-audio-classification#ran","syntology_url":"https://syntology.ai/paper/2204.02121","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02121"}},"official":{"repos":["cheggan/metaaudio-a-few-shot-audio-classification-benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/underwater-single-channel-acoustic-signal","slug":"underwater-single-channel-acoustic-signal","title":"Underwater single-channel acoustic signal multitarget recognition using convolutional neural networks","date":"2022-03-31","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/septr-separable-transformer-for-audio","slug":"septr-separable-transformer-for-audio","title":"SepTr: Separable Transformer for Audio Spectrogram Processing","date":"2022-03-17","arxiv_id":"2203.09581","repositories_listed":1,"syntology":null},{"url":"/paper/auco-resnet-an-end-to-end-network-for-covid","slug":"auco-resnet-an-end-to-end-network-for-covid","title":"AUCO ResNet: an end-to-end network for Covid-19 pre-screening from cough and breath","date":"2022-03-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/hts-at-a-hierarchical-token-semantic-audio","slug":"hts-at-a-hierarchical-token-semantic-audio","title":"HTS-AT: A Hierarchical Token-Semantic Audio Transformer for Sound Classification and Detection","date":"2022-02-02","arxiv_id":"2202.00874","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hts-at-a-hierarchical-token-semantic-audio#ran","syntology_url":"https://syntology.ai/paper/2202.00874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.00874"}},"official":{"repos":["retrocirce/hts-audio-transformer"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/continual-transformers-redundancy-free","slug":"continual-transformers-redundancy-free","title":"Continual Transformers: Redundancy-Free Attention for Online Inference","date":"2022-01-17","arxiv_id":"2201.06268","repositories_listed":1,"syntology":null},{"url":"/paper/classification-of-long-sequential-data-using","slug":"classification-of-long-sequential-data-using","title":"Classification of Long Sequential Data using Circular Dilated Convolutional Neural Networks","date":"2022-01-06","arxiv_id":"2201.02143","repositories_listed":1,"syntology":null},{"url":"/paper/connecting-the-dots-between-audio-and-text","slug":"connecting-the-dots-between-audio-and-text","title":"Connecting the Dots between Audio and Text without Parallel Data through Visual Knowledge Transfer","date":"2021-12-16","arxiv_id":"2112.08995","repositories_listed":1,"syntology":null},{"url":"/paper/x-vector-based-voice-activity-detection-for-1","slug":"x-vector-based-voice-activity-detection-for-1","title":"X-Vector based voice activity detection for multi-genre broadcast speech-to-text","date":"2021-12-09","arxiv_id":"2112.05016","repositories_listed":1,"syntology":null},{"url":"/paper/learning-music-audio-representations-via-weak","slug":"learning-music-audio-representations-via-weak","title":"Learning music audio representations via weak language supervision","date":"2021-12-08","arxiv_id":"2112.04214","repositories_listed":1,"syntology":null},{"url":"/paper/sound-guided-semantic-image-manipulation","slug":"sound-guided-semantic-image-manipulation","title":"Sound-Guided Semantic Image Manipulation","date":"2021-11-30","arxiv_id":"2112.00007","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-audio-visual-representation","slug":"self-supervised-audio-visual-representation","title":"Self-Supervised Audio-Visual Representation Learning with Relaxed Cross-Modal Synchronicity","date":"2021-11-09","arxiv_id":"2111.05329","repositories_listed":1,"syntology":null},{"url":"/paper/lungattn-advanced-lung-sound-classification","slug":"lungattn-advanced-lung-sound-classification","title":"LungAttn: advanced lung sound classification using attention mechanism with dual TQWT and triple STFT spectrogram","date":"2021-10-29","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/study-of-positional-encoding-approaches-for","slug":"study-of-positional-encoding-approaches-for","title":"Study of positional encoding approaches for Audio Spectrogram Transformers","date":"2021-10-13","arxiv_id":"2110.06999","repositories_listed":1,"syntology":null},{"url":"/paper/rank-based-loss-for-learning-hierarchical","slug":"rank-based-loss-for-learning-hierarchical","title":"Rank-based loss for learning hierarchical representations","date":"2021-10-11","arxiv_id":"2110.05941","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rank-based-loss-for-learning-hierarchical#ran","syntology_url":"https://syntology.ai/paper/2110.05941","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.05941"}},"official":{"repos":["inesnolas/Rank-based-loss_ICASSP22"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pruning-vs-xnor-net-a-comprehensive-study-on","slug":"pruning-vs-xnor-net-a-comprehensive-study-on","title":"Pruning vs XNOR-Net: A Comprehensive Study of Deep Learning for Audio Classification on Edge-devices","date":"2021-08-13","arxiv_id":"2108.06128","repositories_listed":1,"syntology":null},{"url":"/paper/sparse-continuous-distributions-and-fenchel","slug":"sparse-continuous-distributions-and-fenchel","title":"Sparse Continuous Distributions and Fenchel-Young Losses","date":"2021-08-04","arxiv_id":"2108.01988","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sparse-continuous-distributions-and-fenchel#ran","syntology_url":"https://syntology.ai/paper/2108.01988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.01988"}},"official":{"repos":["deep-spin/sparse_continuous_distributions"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-veracity-of-local-model-agnostic","slug":"on-the-veracity-of-local-model-agnostic","title":"On the Veracity of Local, Model-agnostic Explanations in Audio Classification: Targeted Investigations with Adversarial Examples","date":"2021-07-19","arxiv_id":"2107.09045","repositories_listed":1,"syntology":null},{"url":"/paper/federated-self-training-for-semi-supervised","slug":"federated-self-training-for-semi-supervised","title":"Federated Self-Training for Semi-Supervised Audio Recognition","date":"2021-07-14","arxiv_id":"2107.06877","repositories_listed":1,"syntology":null},{"url":"/paper/weakly-supervised-classification-and-1","slug":"weakly-supervised-classification-and-1","title":"Weakly-Supervised Classification and Detection of Bird Sounds in the Wild.","date":"2021-07-10","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/attention-bottlenecks-for-multimodal-fusion","slug":"attention-bottlenecks-for-multimodal-fusion","title":"Attention Bottlenecks for Multimodal Fusion","date":"2021-06-30","arxiv_id":"2107.00135","repositories_listed":1,"syntology":null},{"url":"/paper/dcase-2021-task-3-spectrotemporally-aligned","slug":"dcase-2021-task-3-spectrotemporally-aligned","title":"DCASE 2021 Task 3: Spectrotemporally-aligned Features for Polyphonic Sound Event Localization and Detection","date":"2021-06-29","arxiv_id":"2106.15190","repositories_listed":1,"syntology":null},{"url":"/paper/receptive-field-regularization-techniques-for","slug":"receptive-field-regularization-techniques-for","title":"Receptive Field Regularization Techniques for Audio Classification and Tagging with Deep Convolutional Neural Networks","date":"2021-05-26","arxiv_id":"2105.12395","repositories_listed":1,"syntology":null},{"url":"/paper/sparse-spiking-gradient-descent","slug":"sparse-spiking-gradient-descent","title":"Sparse Spiking Gradient Descent","date":"2021-05-18","arxiv_id":"2105.08810","repositories_listed":1,"syntology":null},{"url":"/paper/broaden-your-views-for-self-supervised-video","slug":"broaden-your-views-for-self-supervised-video","title":"Broaden Your Views for Self-Supervised Video Learning","date":"2021-03-30","arxiv_id":"2103.16559","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/broaden-your-views-for-self-supervised-video#ran","syntology_url":"https://syntology.ai/paper/2103.16559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.16559"}},"official":{"repos":["deepmind/brave"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/environmental-sound-classification-on-the","slug":"environmental-sound-classification-on-the","title":"Environmental Sound Classification on the Edge: A Pipeline for Deep Acoustic Networks on Extremely Resource-Constrained Devices","date":"2021-03-05","arxiv_id":"2103.03483","repositories_listed":1,"syntology":null},{"url":"/paper/improving-deep-learning-based-semi-supervised","slug":"improving-deep-learning-based-semi-supervised","title":"Comparison of semi-supervised deep learning algorithms for audio classification","date":"2021-02-16","arxiv_id":"2102.08183","repositories_listed":1,"syntology":null},{"url":"/paper/psla-improving-audio-event-classification","slug":"psla-improving-audio-event-classification","title":"PSLA: Improving Audio Tagging with Pretraining, Sampling, Labeling, and Aggregation","date":"2021-02-02","arxiv_id":"2102.01243","repositories_listed":1,"syntology":null},{"url":"/paper/piano-skills-assessment","slug":"piano-skills-assessment","title":"Piano Skills Assessment","date":"2021-01-13","arxiv_id":"2101.04884","repositories_listed":1,"syntology":null},{"url":"/paper/respirenet-a-deep-neural-network-for","slug":"respirenet-a-deep-neural-network-for","title":"RespireNet: A Deep Neural Network for Accurately Detecting Abnormal Lung Sounds in Limited Data Setting","date":"2020-10-31","arxiv_id":"2011.00196","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/respirenet-a-deep-neural-network-for#ran","syntology_url":"https://syntology.ai/paper/2011.00196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.00196"}},"official":{"repos":["microsoft/RespireNet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/urban-sound-classification-striving-towards-a","slug":"urban-sound-classification-striving-towards-a","title":"Urban Sound Classification : striving towards a fair comparison","date":"2020-10-22","arxiv_id":"2010.11805","repositories_listed":1,"syntology":null},{"url":"/paper/ultra-light-deep-mir-by-trimming-lottery","slug":"ultra-light-deep-mir-by-trimming-lottery","title":"Ultra-light deep MIR by trimming lottery tickets","date":"2020-07-31","arxiv_id":"2007.16187","repositories_listed":1,"syntology":null},{"url":"/paper/text-based-classification-of-interviews-for","slug":"text-based-classification-of-interviews-for","title":"Text-based classification of interviews for mental health -- juxtaposing the state of the art","date":"2020-07-29","arxiv_id":"2008.01543","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-multimodal-versatile-networks","slug":"self-supervised-multimodal-versatile-networks","title":"Self-Supervised MultiModal Versatile Networks","date":"2020-06-29","arxiv_id":"2006.16228","repositories_listed":1,"syntology":null},{"url":"/paper/audio-visual-instance-discrimination-with","slug":"audio-visual-instance-discrimination-with","title":"Audio-Visual Instance Discrimination with Cross-Modal Agreement","date":"2020-04-27","arxiv_id":"2004.12943","repositories_listed":1,"syntology":null},{"url":"/paper/multi-modal-self-supervision-from-generalized","slug":"multi-modal-self-supervision-from-generalized","title":"On Compositions of Transformations in Contrastive Self-Supervised Learning","date":"2020-03-09","arxiv_id":"2003.04298","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-modal-self-supervision-from-generalized#ran","syntology_url":"https://syntology.ai/paper/2003.04298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.04298"}},"official":{"repos":["facebookresearch/GDT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/speech-emotion-recognition-with-deep","slug":"speech-emotion-recognition-with-deep","title":"Speech emotion recognition with deep convolutional neural networks","date":"2020-02-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-with-out-of-distribution-data-for","slug":"learning-with-out-of-distribution-data-for","title":"Learning with Out-of-Distribution Data for Audio Classification","date":"2020-02-11","arxiv_id":"2002.04683","repositories_listed":1,"syntology":null},{"url":"/paper/audiogmenter-a-matlab-toolbox-for-audio-data","slug":"audiogmenter-a-matlab-toolbox-for-audio-data","title":"Audiogmenter: a MATLAB Toolbox for Audio Data Augmentation","date":"2019-12-11","arxiv_id":"1912.05472","repositories_listed":1,"syntology":null},{"url":"/paper/lungbrn-a-smart-digital-stethoscope-for","slug":"lungbrn-a-smart-digital-stethoscope-for","title":"LungBRN: A Smart Digital Stethoscope for Detecting Respiratory Disease Using bi-ResNet Deep Learning Algorithm","date":"2019-12-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-learning-by-cross-modal-audio","slug":"self-supervised-learning-by-cross-modal-audio","title":"Self-Supervised Learning by Cross-Modal Audio-Video Clustering","date":"2019-11-28","arxiv_id":"1911.12667","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/self-supervised-learning-by-cross-modal-audio#ran","syntology_url":"https://syntology.ai/paper/1911.12667","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.12667"}},"official":{"repos":["HumamAlwassel/XDC"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/universal-adversarial-audio-perturbations","slug":"universal-adversarial-audio-perturbations","title":"Universal Adversarial Audio Perturbations","date":"2019-08-08","arxiv_id":"1908.03173","repositories_listed":1,"syntology":null},{"url":"/paper/compact-global-descriptor-for-neural-networks","slug":"compact-global-descriptor-for-neural-networks","title":"Compact Global Descriptor for Neural Networks","date":"2019-07-23","arxiv_id":"1907.09665","repositories_listed":1,"syntology":null},{"url":"/paper/audio-visual-model-distillation-using","slug":"audio-visual-model-distillation-using","title":"Audio-Visual Model Distillation Using Acoustic Images","date":"2019-04-16","arxiv_id":"1904.07933","repositories_listed":1,"syntology":null},{"url":"/paper/ubicoustics-plug-and-play-acoustic-activity","slug":"ubicoustics-plug-and-play-acoustic-activity","title":"Ubicoustics: Plug-and-Play Acoustic Activity Recognition","date":"2018-10-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-closer-look-at-weak-label-learning-for","slug":"a-closer-look-at-weak-label-learning-for","title":"A Closer Look at Weak Label Learning for Audio Events","date":"2018-04-24","arxiv_id":"1804.09288","repositories_listed":1,"syntology":null},{"url":"/paper/masked-conditional-neural-networks-for-audio","slug":"masked-conditional-neural-networks-for-audio","title":"Masked Conditional Neural Networks for Audio Classification","date":"2018-03-06","arxiv_id":"1803.02421","repositories_listed":1,"syntology":null},{"url":"/paper/look-listen-and-learn","slug":"look-listen-and-learn","title":"Look, Listen and Learn","date":"2017-05-23","arxiv_id":"1705.08168","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-bag-of-features-model-for-music-auto","slug":"a-deep-bag-of-features-model-for-music-auto","title":"A Deep Bag-of-Features Model for Music Auto-Tagging","date":"2015-08-20","arxiv_id":"1508.04999","repositories_listed":1,"syntology":null},{"url":null,"slug":"mupax-multidimensional-problem-agnostic","title":"MUPAX: Multidimensional Problem Agnostic eXplainable AI","date":"2025-07-17","arxiv_id":"2507.13090","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-specific-audio-coding-for-machines","title":"Task-Specific Audio Coding for Machines: Machine-Learned Latent Features Are Codes for That Machine","date":"2025-07-17","arxiv_id":"2507.12701","repositories_listed":0,"syntology":null},{"url":null,"slug":"neuromorphic-wireless-split-computing-with-1","title":"Neuromorphic Wireless Split Computing with Resonate-and-Fire Neurons","date":"2025-06-24","arxiv_id":"2506.20015","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-implicitly-learn-to-see","title":"Large Language Models Implicitly Learn to See and Hear Just By Reading","date":"2025-05-20","arxiv_id":"2505.17091","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-masked-autoencoders-also-listen-to-birds","title":"Can Masked Autoencoders Also Listen to Birds?","date":"2025-04-17","arxiv_id":"2504.12880","repositories_listed":0,"syntology":null},{"url":null,"slug":"progressive-rock-music-classification","title":"Progressive Rock Music Classification","date":"2025-04-15","arxiv_id":"2504.10821","repositories_listed":0,"syntology":null},{"url":"/paper/ca-2st-cross-attention-in-audio-space-and","slug":"ca-2st-cross-attention-in-audio-space-and","title":"CA^2ST: Cross-Attention in Audio, Space, and Time for Holistic Video Recognition","date":"2025-03-30","arxiv_id":"2503.23447","repositories_listed":0,"syntology":null},{"url":null,"slug":"symbolic-audio-classification-via-modal","title":"Symbolic Audio Classification via Modal Decision Tree Learning","date":"2025-03-21","arxiv_id":"2503.17018","repositories_listed":0,"syntology":null},{"url":null,"slug":"fundamental-survey-on-neuromorphic-based","title":"Fundamental Survey on Neuromorphic Based Audio Classification","date":"2025-02-20","arxiv_id":"2502.15056","repositories_listed":0,"syntology":null},{"url":"/paper/lhgnn-local-higher-order-graph-neural","slug":"lhgnn-local-higher-order-graph-neural","title":"LHGNN: Local-Higher Order Graph Neural Networks For Audio Classification and Tagging","date":"2025-01-07","arxiv_id":"2501.03464","repositories_listed":0,"syntology":null},{"url":null,"slug":"tspe-task-specific-prompt-ensemble-for","title":"TSPE: Task-Specific Prompt Ensemble for Improved Zero-Shot Audio Classification","date":"2024-12-31","arxiv_id":"2501.00398","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-incremental-learning-for-audio","title":"Domain-Incremental Learning for Audio Classification","date":"2024-12-23","arxiv_id":"2412.17424","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-low-rank-scaled-dot-product","title":"Continual Low-Rank Scaled Dot-product Attention","date":"2024-12-04","arxiv_id":"2412.03214","repositories_listed":0,"syntology":null},{"url":null,"slug":"raw-audio-classification-with-cosine","title":"Raw Audio Classification with Cosine Convolutional Neural Network (CosCovNN)","date":"2024-11-30","arxiv_id":"2412.00312","repositories_listed":0,"syntology":null},{"url":null,"slug":"stream-a-universal-state-space-model-for","title":"STREAM: A Universal State-Space Model for Sparse Geometric Data","date":"2024-11-19","arxiv_id":"2411.12603","repositories_listed":0,"syntology":null},{"url":null,"slug":"classification-of-adventitious-sounds","title":"Classification of Adventitious Sounds Combining Cochleogram and Vision Transformers","date":"2024-11-08","arxiv_id":"2411.05955","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-score-cam-based-denoiser-for","title":"A Novel Score-CAM based Denoiser for Spectrographic Signature Extraction without Ground Truth","date":"2024-10-28","arxiv_id":"2410.21557","repositories_listed":0,"syntology":null}],"record_sha256":"b4d021c6125389cb59594c6e40235b6a6e1dc6f1bfed1482a1fe50f72383755a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}