{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/dataset/audioset/papers/ran/1","list_of":"/dataset/audioset","dataset":"AudioSet","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","key_notes":{"samples_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","samples_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this dataset or check it against this dataset's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","population":"every paper with a leaderboard row on this dataset's benchmarks (the benchmark-backed subset): the archive's own papers-using-this-dataset list was never published, so this is not that list; num_papers_in_archive is the archive's own count","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,16],"of":16,"counts":{"papers_with_a_benchmark_row":38,"with_a_code_link":31,"where_syntology_ran_a_sample":16,"not_listed_spam_title":0,"listed":38,"listed_where_code_ran":16,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":15,"every_run_a_failure_of_syntologys_instrument":1,"listed_with_a_run_with_no_instrument_failure":15,"listed_every_run_a_failure_of_syntologys_instrument":1,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers with at least one leaderboard row on this dataset's benchmarks; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/dataset/audioset/papers/ran/1","prev":null,"next":null,"papers":[{"paper":"/paper/sslam-enhancing-self-supervised-models-with-1","slug":"sslam-enhancing-self-supervised-models-with-1","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","date":"2025-06-13","arxiv_id":"2506.12222","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":1,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/sslam-enhancing-self-supervised-models-with-1#ran","syntology_url":"https://syntology.ai/paper/2506.12222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.12222"}}}},{"paper":"/paper/dass-distilled-audio-state-space-models-are","slug":"dass-distilled-audio-state-space-models-are","title":"DASS: Distilled Audio State Space Models Are Stronger and More Duration-Scalable Learners","date":"2024-07-04","arxiv_id":"2407.04082","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["Saurabhbhati/DASS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/dass-distilled-audio-state-space-models-are#ran","syntology_url":"https://syntology.ai/paper/2407.04082","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04082"}}}},{"paper":"/paper/masked-modeling-duo-towards-a-universal-audio","slug":"masked-modeling-duo-towards-a-universal-audio","title":"Masked Modeling Duo: Towards a Universal Audio Pre-training Framework","date":"2024-04-09","arxiv_id":"2404.06095","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":7,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":7,"official":{"repos":["nttcslab/m2d","nttcslab/eval-audio-repr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/masked-modeling-duo-towards-a-universal-audio#ran","syntology_url":"https://syntology.ai/paper/2404.06095","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.06095"}}}},{"paper":"/paper/equiav-leveraging-equivariance-for-audio","slug":"equiav-leveraging-equivariance-for-audio","title":"EquiAV: Leveraging Equivariance for Audio-Visual Contrastive Learning","date":"2024-03-14","arxiv_id":"2403.09502","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":13,"samples_ran":6,"samples_constructed":4,"samples_ran_checked":6,"samples_ran_instrument_failed":0,"samples_unverified":7,"pointer_only_for_licence":2,"official":{"repos":["jongsuk1/equiav"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":4,"n_ran_no_instrument_failure":6,"n_unverified":7,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/equiav-leveraging-equivariance-for-audio#ran","syntology_url":"https://syntology.ai/paper/2403.09502","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09502"}}}},{"paper":"/paper/clapsep-leveraging-contrastive-pre-trained","slug":"clapsep-leveraging-contrastive-pre-trained","title":"CLAPSep: Leveraging Contrastive Pre-trained Model for Multi-Modal Query-Conditioned Target Sound Extraction","date":"2024-02-27","arxiv_id":"2402.17455","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["aisaka0v0/clapsep"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/clapsep-leveraging-contrastive-pre-trained#ran","syntology_url":"https://syntology.ai/paper/2402.17455","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17455"}}}},{"paper":"/paper/eat-self-supervised-pre-training-with","slug":"eat-self-supervised-pre-training-with","title":"EAT: Self-Supervised Pre-Training with Efficient Audio Transformer","date":"2024-01-07","arxiv_id":"2401.03497","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":17,"samples_ran":13,"samples_constructed":0,"samples_ran_checked":13,"samples_ran_instrument_failed":0,"samples_unverified":4,"pointer_only_for_licence":3,"official":{"repos":["cwx-worst-one/eat"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":4,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/eat-self-supervised-pre-training-with#ran","syntology_url":"https://syntology.ai/paper/2401.03497","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03497"}}}},{"paper":"/paper/beats-audio-pre-training-with-acoustic","slug":"beats-audio-pre-training-with-acoustic","title":"BEATs: Audio Pre-Training with Acoustic Tokenizers","date":"2022-12-18","arxiv_id":"2212.09058","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":18,"samples_ran":16,"samples_constructed":0,"samples_ran_checked":13,"samples_ran_instrument_failed":3,"samples_unverified":2,"pointer_only_for_licence":4,"official":{"repos":["microsoft/unilm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/beats-audio-pre-training-with-acoustic#ran","syntology_url":"https://syntology.ai/paper/2212.09058","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09058"}}}},{"paper":"/paper/efficient-large-scale-audio-tagging-via","slug":"efficient-large-scale-audio-tagging-via","title":"Efficient Large-scale Audio Tagging via Transformer-to-CNN Knowledge Distillation","date":"2022-11-09","arxiv_id":"2211.04772","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["fschmid56/efficientat"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/efficient-large-scale-audio-tagging-via#ran","syntology_url":"https://syntology.ai/paper/2211.04772","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.04772"}}}},{"paper":"/paper/uavm-a-unified-model-for-audio-visual","slug":"uavm-a-unified-model-for-audio-visual","title":"UAVM: Towards Unifying Audio and Visual Models","date":"2022-07-29","arxiv_id":"2208.00061","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":14,"samples_ran":10,"samples_constructed":0,"samples_ran_checked":10,"samples_ran_instrument_failed":0,"samples_unverified":4,"pointer_only_for_licence":3,"official":{"repos":["YuanGongND/uavm"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/uavm-a-unified-model-for-audio-visual#ran","syntology_url":"https://syntology.ai/paper/2208.00061","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.00061"}}}},{"paper":"/paper/hts-at-a-hierarchical-token-semantic-audio","slug":"hts-at-a-hierarchical-token-semantic-audio","title":"HTS-AT: A Hierarchical Token-Semantic Audio Transformer for Sound Classification and Detection","date":"2022-02-02","arxiv_id":"2202.00874","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":9,"samples_ran":9,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":4,"samples_unverified":0,"pointer_only_for_licence":3,"official":{"repos":["retrocirce/hts-audio-transformer"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/hts-at-a-hierarchical-token-semantic-audio#ran","syntology_url":"https://syntology.ai/paper/2202.00874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.00874"}}}},{"paper":"/paper/zero-shot-audio-source-separation-through","slug":"zero-shot-audio-source-separation-through","title":"Zero-shot Audio Source Separation through Query-based Learning from Weakly-labeled Data","date":"2021-12-15","arxiv_id":"2112.07891","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":10,"samples_ran":9,"samples_constructed":0,"samples_ran_checked":6,"samples_ran_instrument_failed":3,"samples_unverified":1,"pointer_only_for_licence":3,"official":{"repos":["RetroCirce/Zero_Shot_Audio_Source_Separation"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/zero-shot-audio-source-separation-through#ran","syntology_url":"https://syntology.ai/paper/2112.07891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.07891"}}}},{"paper":"/paper/efficient-training-of-audio-transformers-with","slug":"efficient-training-of-audio-transformers-with","title":"Efficient Training of Audio Transformers with Patchout","date":"2021-10-11","arxiv_id":"2110.05069","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":3,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["kkoutini/passt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/efficient-training-of-audio-transformers-with#ran","syntology_url":"https://syntology.ai/paper/2110.05069","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.05069"}}}},{"paper":"/paper/vatt-transformers-for-multimodal-self","slug":"vatt-transformers-for-multimodal-self","title":"VATT: Transformers for Multimodal Self-Supervised Learning from Raw Video, Audio and Text","date":"2021-04-22","arxiv_id":"2104.11178","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":5,"samples_constructed":4,"samples_ran_checked":5,"samples_ran_instrument_failed":0,"samples_unverified":3,"pointer_only_for_licence":8,"official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/vatt-transformers-for-multimodal-self#ran","syntology_url":"https://syntology.ai/paper/2104.11178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.11178"}}}},{"paper":"/paper/ast-audio-spectrogram-transformer","slug":"ast-audio-spectrogram-transformer","title":"AST: Audio Spectrogram Transformer","date":"2021-04-05","arxiv_id":"2104.01778","rows_on_this_dataset":3,"code_links":5,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["YuanGongND/ast"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/ast-audio-spectrogram-transformer#ran","syntology_url":"https://syntology.ai/paper/2104.01778","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.01778"}}}},{"paper":"/paper/perceiver-general-perception-with-iterative","slug":"perceiver-general-perception-with-iterative","title":"Perceiver: General Perception with Iterative Attention","date":"2021-03-04","arxiv_id":"2103.03206","rows_on_this_dataset":1,"code_links":12,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":55,"samples_ran":41,"samples_constructed":20,"samples_ran_checked":30,"samples_ran_instrument_failed":11,"samples_unverified":14,"pointer_only_for_licence":11,"official":{"repos":["deepmind/deepmind-research"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["community","listed","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/perceiver-general-perception-with-iterative#ran","syntology_url":"https://syntology.ai/paper/2103.03206","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.03206"}}}},{"paper":"/paper/panns-large-scale-pretrained-audio-neural-1","slug":"panns-large-scale-pretrained-audio-neural-1","title":"PANNs: Large-Scale Pretrained Audio Neural Networks for Audio Pattern Recognition","date":null,"arxiv_id":"1912.10211","rows_on_this_dataset":2,"code_links":8,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":16,"samples_ran":14,"samples_constructed":0,"samples_ran_checked":14,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["yinkalario/General-Purpose-Sound-Recognition-Demo","qiuqiangkong/audioset_tagging_cnn"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/panns-large-scale-pretrained-audio-neural-1#ran","syntology_url":"https://syntology.ai/paper/1912.10211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.10211"}}}}],"record_sha256":"37dd6e74d64d3e9b9a6a3d69175761f0e17f91d5d8d54c2c266c3a7c1cd41aaa","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}