{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/av-superb-a-multi-task-evaluation-benchmark","title":"AV-SUPERB: A Multi-Task Evaluation Benchmark for Audio-Visual Representation Models","arxiv_id":"2309.10787","date":"2023-09-19","proceeding":null,"authors":["Yuan Tseng","Layne Berry","Yi-Ting Chen","I-Hsiang Chiu","Hsuan-Hao Lin","Max Liu","Puyuan Peng","Yi-Jen Shih","Hung-Yu Wang","Haibin Wu","Po-Yao Huang","Chun-Mao Lai","Shang-Wen Li","David Harwath","Yu Tsao","Shinji Watanabe","Abdelrahman Mohamed","Chi-Luen Feng","Hung-Yi Lee"],"abstract":"Audio-visual representation learning aims to develop systems with human-like perception by utilizing correlation between auditory and visual information. However, current models often focus on a limited set of tasks, and generalization abilities of learned representations are unclear. To this end, we propose the AV-SUPERB benchmark that enables general-purpose evaluation of unimodal audio/visual and bimodal fusion representations on 7 datasets covering 5 audio-visual tasks in speech and audio processing. We evaluate 5 recent self-supervised models and show that none of these models generalize to all tasks, emphasizing the need for future study on improving universal model performance. In addition, we show that representations may be improved with intermediate-task fine-tuning and audio event classification with AudioSet serves as a strong intermediate task. We release our benchmark with evaluation code and a model submission platform to encourage further research in audio-visual learning.","url_abs":"https://arxiv.org/abs/2309.10787v2","url_pdf":"https://arxiv.org/pdf/2309.10787v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"av-superb-a-multi-task-evaluation-benchmark","repo_url":"https://github.com/roger-tseng/av-superb","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"NOASSERTION"}}],"tasks":[{"task_slug":"representation-learning","task_name":"Representation Learning"},{"task_slug":"audio-visual-learning","task_name":"audio-visual learning"}],"methods":[{"method_slug":"focus","method_name":"Focus"},{"method_slug":null,"method_name":"None"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2309.10787","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.10787"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/roger-tseng/av-superb","reach":{"status":"ok","spdx":"NOASSERTION"}}],"summary":{"ran":2,"unverified":1},"by_repo_kind":{"official":{"samples":3,"ran":2,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":3,"samples":[{"code_sha256_prefix":"aeee09ff82b8dc3c","entry":"decide_utter_input_dim","repo":"roger-tseng/av-superb","repo_kind":"official","path":"downstream_tasks/sv/model.py","file_url":"https://github.com/roger-tseng/av-superb/blob/HEAD/downstream_tasks/sv/model.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"aeee09ff82b8dc3c"}},{"code_sha256_prefix":"e27f21a0984c5fa5","entry":"options","repo":"roger-tseng/av-superb","repo_kind":"official","path":"hub.py","file_url":"https://github.com/roger-tseng/av-superb/blob/HEAD/hub.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"e27f21a0984c5fa5"}},{"code_sha256_prefix":"8c3953350394e908","entry":"get_ddp_sampler","repo":"roger-tseng/av-superb","repo_kind":"official","path":"downstream_tasks/kinetics_sounds/expert.py","file_url":"https://github.com/roger-tseng/av-superb/blob/HEAD/downstream_tasks/kinetics_sounds/expert.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"8c3953350394e908"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}