{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/2506-08967","title":"Step-Audio-AQAA: a Fully End-to-End Expressive Large Audio Language Model","arxiv_id":"2506.08967","date":"2025-06-10","proceeding":null,"authors":["Ailin Huang","Bingxin Li","Bruce Wang","Boyong Wu","Chao Yan","Chengli Feng","Heng Wang","HongYu Zhou","Hongyuan Wang","Jingbei Li","Jianjian Sun","Joanna Wang","Mingrui Chen","Peng Liu","Ruihang Miao","Shilei Jiang","Tian Fei","Wang You","Xi Chen","Xuerui Yang","Yechang Huang","Yuxiang Zhang","Zheng Ge","Zheng Gong","Zhewei Huang","Zixin Zhang","Bin Wang","Bo Li","Buyun Ma","Changxin Miao","Changyi Wan","Chen Xu","Dapeng Shi","Dingyuan Hu","Enle Liu","Guanzhe Huang","Gulin Yan","Hanpeng Hu","Haonan Jia","Jiahao Gong","Jiaoren Wu","Jie Wu","Jie Yang","Junzhe Lin","Kaixiang Li","Lei Xia","Longlong Gu","Ming Li","Nie Hao","Ranchen Ming","Shaoliang Pang","SiQi Liu","Song Yuan","Tiancheng Cao","Wen Li","Wenqing He","Xu Zhao","Xuelin Zhang","Yanbo Yu","Yinmin Zhong","Yu Zhou","Yuanwei Liang","Yuanwei Lu","Yuxiang Yang","Zidong Yang","Zili Zhang","Binxing Jiao","Heung-Yeung Shum","Jiansheng Chen","Jing Li","Xiangyu Zhang","Xinhao Zhang","Yibo Zhu","Daxin Jiang","Shuchang Zhou","Chen Hu"],"abstract":"Large Audio-Language Models (LALMs) have significantly advanced intelligent human-computer interaction, yet their reliance on text-based outputs limits their ability to generate natural speech responses directly, hindering seamless audio interactions. To address this, we introduce Step-Audio-AQAA, a fully end-to-end LALM designed for Audio Query-Audio Answer (AQAA) tasks. The model integrates a dual-codebook audio tokenizer for linguistic and semantic feature extraction, a 130-billion-parameter backbone LLM and a neural vocoder for high-fidelity speech synthesis. Our post-training approach employs interleaved token-output of text and audio to enhance semantic coherence and combines Direct Preference Optimization (DPO) with model merge to improve performance. Evaluations on the StepEval-Audio-360 benchmark demonstrate that Step-Audio-AQAA excels especially in speech control, outperforming the state-of-art LALMs in key areas. This work contributes a promising solution for end-to-end LALMs and highlights the critical role of token-based vocoder in enhancing overall performance for AQAA tasks.","url_abs":"https://arxiv.org/abs/2506.08967v2","url_pdf":"https://arxiv.org/pdf/2506.08967v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"2506-08967","repo_url":"https://github.com/stepfun-ai/step-audio","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}}],"tasks":[{"task_slug":"language-modeling","task_name":"Language Modeling"},{"task_slug":"language-modelling","task_name":"Language Modelling"},{"task_slug":"speech-synthesis","task_name":"Speech Synthesis"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2506.08967","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08967"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/stepfun-ai/step-audio","reach":{"status":"ok","spdx":"Apache-2.0"}}],"summary":{"ran":1,"unverified":5},"by_repo_kind":{"listed":{"samples":6,"ran":1,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"8043eaefe4c09c1f","entry":"load_wav","repo":"stepfun-ai/step-audio","repo_kind":"listed","path":"cosyvoice/matcha/audio.py","file_url":"https://github.com/stepfun-ai/step-audio/blob/HEAD/cosyvoice/matcha/audio.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"8043eaefe4c09c1f"}},{"code_sha256_prefix":"be031b15a988a556","entry":"dynamic_range_compression","repo":"stepfun-ai/step-audio","repo_kind":"listed","path":"cosyvoice/matcha/audio.py","file_url":"https://github.com/stepfun-ai/step-audio/blob/HEAD/cosyvoice/matcha/audio.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"be031b15a988a556"}},{"code_sha256_prefix":"9e30e7d510c9d341","entry":"dynamic_range_decompression","repo":"stepfun-ai/step-audio","repo_kind":"listed","path":"cosyvoice/matcha/audio.py","file_url":"https://github.com/stepfun-ai/step-audio/blob/HEAD/cosyvoice/matcha/audio.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"9e30e7d510c9d341"}},{"code_sha256_prefix":"1e0fb79d41422266","entry":"load_audio","repo":"stepfun-ai/step-audio","repo_kind":"listed","path":"funasr_detach/models/whisper/utils/audio.py","file_url":"https://github.com/stepfun-ai/step-audio/blob/HEAD/funasr_detach/models/whisper/utils/audio.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"1e0fb79d41422266"}},{"code_sha256_prefix":"edf924293b4b2375","entry":"mel_filters","repo":"stepfun-ai/step-audio","repo_kind":"listed","path":"funasr_detach/models/whisper/utils/audio.py","file_url":"https://github.com/stepfun-ai/step-audio/blob/HEAD/funasr_detach/models/whisper/utils/audio.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"edf924293b4b2375"}},{"code_sha256_prefix":"9f13de1c23e4003d","entry":"pad_or_trim","repo":"stepfun-ai/step-audio","repo_kind":"listed","path":"funasr_detach/models/whisper/utils/audio.py","file_url":"https://github.com/stepfun-ai/step-audio/blob/HEAD/funasr_detach/models/whisper/utils/audio.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"9f13de1c23e4003d"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}