{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/glap-general-contrastive-audio-text","title":"GLAP: General contrastive audio-text pretraining across domains and languages","arxiv_id":"2506.11350","date":"2025-06-12","proceeding":null,"authors":["Heinrich Dinkel","Zhiyong Yan","Tianzi Wang","Yongqing Wang","Xingwei Sun","Yadong Niu","Jizhong Liu","Gang Li","Junbo Zhang","Jian Luan"],"abstract":"Contrastive Language Audio Pretraining (CLAP) is a widely-used method to bridge the gap between audio and text domains. Current CLAP methods enable sound and music retrieval in English, ignoring multilingual spoken content. To address this, we introduce general language audio pretraining (GLAP), which expands CLAP with multilingual and multi-domain abilities. GLAP demonstrates its versatility by achieving competitive performance on standard audio-text retrieval benchmarks like Clotho and AudioCaps, while significantly surpassing existing methods in speech retrieval and classification tasks. Additionally, GLAP achieves strong results on widely used sound-event zero-shot benchmarks, while simultaneously outperforming previous methods on speech content benchmarks. Further keyword spotting evaluations across 50 languages emphasize GLAP's advanced multilingual capabilities. Finally, multilingual sound and music understanding is evaluated across four languages. Checkpoints and Source: https://github.com/xiaomi-research/dasheng-glap.","url_abs":"https://arxiv.org/abs/2506.11350v1","url_pdf":"https://arxiv.org/pdf/2506.11350v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"glap-general-contrastive-audio-text","repo_url":"https://github.com/xiaomi-research/dasheng-glap","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"jax","reach":{"status":"ok","spdx":"Apache-2.0"}}],"tasks":[{"task_slug":null,"task_name":"AudioCaps"},{"task_slug":"keyword-spotting","task_name":"Keyword Spotting"},{"task_slug":"retrieval","task_name":"Retrieval"},{"task_slug":"text-retrieval","task_name":"Text Retrieval"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2506.11350","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.11350"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/xiaomi-research/dasheng-glap","reach":{"status":"ok","spdx":"Apache-2.0"}}],"summary":{"ran":1,"unverified":9},"by_repo_kind":{"official":{"samples":10,"ran":1,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"7f1946b9c385a5cf","entry":"cached","repo":"xiaomi-research/dasheng-glap","repo_kind":"official","path":"glap_model/grad_cache/functional.py","file_url":"https://github.com/xiaomi-research/dasheng-glap/blob/HEAD/glap_model/grad_cache/functional.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"7f1946b9c385a5cf"}},{"code_sha256_prefix":"fab5db4632e2fa87","entry":"cat_input_tensor","repo":"xiaomi-research/dasheng-glap","repo_kind":"official","path":"glap_model/grad_cache/functional.py","file_url":"https://github.com/xiaomi-research/dasheng-glap/blob/HEAD/glap_model/grad_cache/functional.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"fab5db4632e2fa87"}},{"code_sha256_prefix":"061cc02ff1e95a46","entry":"decode_torch_audio","repo":"xiaomi-research/dasheng-glap","repo_kind":"official","path":"glap_model/dataset.py","file_url":"https://github.com/xiaomi-research/dasheng-glap/blob/HEAD/glap_model/dataset.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"061cc02ff1e95a46"}},{"code_sha256_prefix":"c4895a0012bf6dfc","entry":"exists","repo":"xiaomi-research/dasheng-glap","repo_kind":"official","path":"glap_model/dataset.py","file_url":"https://github.com/xiaomi-research/dasheng-glap/blob/HEAD/glap_model/dataset.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"c4895a0012bf6dfc"}},{"code_sha256_prefix":"df75b65d469704ea","entry":"fast_warn_and_continue","repo":"xiaomi-research/dasheng-glap","repo_kind":"official","path":"glap_model/dataset.py","file_url":"https://github.com/xiaomi-research/dasheng-glap/blob/HEAD/glap_model/dataset.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"df75b65d469704ea"}},{"code_sha256_prefix":"0ec2f1bb15ab20d3","entry":"freeze_model","repo":"xiaomi-research/dasheng-glap","repo_kind":"official","path":"glap_model/models/glap.py","file_url":"https://github.com/xiaomi-research/dasheng-glap/blob/HEAD/glap_model/models/glap.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"0ec2f1bb15ab20d3"}},{"code_sha256_prefix":"7917a2079ec6b095","entry":"gather_input_tensor","repo":"xiaomi-research/dasheng-glap","repo_kind":"official","path":"glap_model/grad_cache/functional.py","file_url":"https://github.com/xiaomi-research/dasheng-glap/blob/HEAD/glap_model/grad_cache/functional.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"7917a2079ec6b095"}},{"code_sha256_prefix":"0245565a9267ff41","entry":"gather_tensors","repo":"xiaomi-research/dasheng-glap","repo_kind":"official","path":"glap_model/trainer.py","file_url":"https://github.com/xiaomi-research/dasheng-glap/blob/HEAD/glap_model/trainer.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"0245565a9267ff41"}},{"code_sha256_prefix":"10059c78cc1632fe","entry":"load_or_download_pretrained_checkpoint","repo":"xiaomi-research/dasheng-glap","repo_kind":"official","path":"glap_model/models/glap.py","file_url":"https://github.com/xiaomi-research/dasheng-glap/blob/HEAD/glap_model/models/glap.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"10059c78cc1632fe"}},{"code_sha256_prefix":"018ad58b48112aa9","entry":"parse_spectransforms","repo":"xiaomi-research/dasheng-glap","repo_kind":"official","path":"glap_model/models/audioencoders/dasheng_audioencoder.py","file_url":"https://github.com/xiaomi-research/dasheng-glap/blob/HEAD/glap_model/models/audioencoders/dasheng_audioencoder.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"018ad58b48112aa9"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}