{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/audio-flamingo-a-novel-audio-language-model","title":"Audio Flamingo: A Novel Audio Language Model with Few-Shot Learning and Dialogue Abilities","arxiv_id":"2402.01831","date":"2024-02-02","proceeding":null,"authors":["Zhifeng Kong","Arushi Goel","Rohan Badlani","Wei Ping","Rafael Valle","Bryan Catanzaro"],"abstract":"Augmenting large language models (LLMs) to understand audio -- including non-speech sounds and non-verbal speech -- is critically important for diverse real-world applications of LLMs. In this paper, we propose Audio Flamingo, a novel audio language model with 1) strong audio understanding abilities, 2) the ability to quickly adapt to unseen tasks via in-context learning and retrieval, and 3) strong multi-turn dialogue abilities. We introduce a series of training techniques, architecture design, and data strategies to enhance our model with these abilities. Extensive evaluations across various audio understanding tasks confirm the efficacy of our method, setting new state-of-the-art benchmarks. Our demo website is https://audioflamingo.github.io/ and the code is open-sourced at https://github.com/NVIDIA/audio-flamingo.","url_abs":"https://arxiv.org/abs/2402.01831v3","url_pdf":"https://arxiv.org/pdf/2402.01831v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"audio-flamingo-a-novel-audio-language-model","repo_url":"https://github.com/NVIDIA/audio-flamingo","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}}],"tasks":[{"task_slug":"acoustic-scene-classification","task_name":"Acoustic Scene Classification"},{"task_slug":"audio-captioning","task_name":"Audio captioning"},{"task_slug":"few-shot-learning","task_name":"Few-Shot Learning"},{"task_slug":"in-context-learning","task_name":"In-Context Learning"},{"task_slug":"language-modeling","task_name":"Language Modeling"},{"task_slug":"language-modelling","task_name":"Language Modelling"},{"task_slug":"retrieval","task_name":"Retrieval"},{"task_slug":"retrieval-augmented-few-shot-in-context-audio","task_name":"Retrieval-augmented Few-shot In-context Audio Captioning"},{"task_slug":"zero-shot-audio-captioning","task_name":"Zero-shot Audio Captioning"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/acoustic-scene-classification-on-cochlscene","task":"Acoustic Scene Classification","dataset":"CochlScene","model":"Audio Flamingo","rank_in_archive_order":1,"of":2,"metrics":{"1:1 Accuracy":"0.830"},"uses_additional_data":true},{"leaderboard":"/sota/audio-captioning-on-clotho","task":"Audio captioning","dataset":"Clotho","model":"Audio Flamingo (Pengi trainset)","rank_in_archive_order":5,"of":11,"metrics":{"BLEU-4":"17.4","CIDEr":"0.489","METEOR":"18.7","ROUGE-L":"39.4","SPICE":"0.134","SPIDEr":"0.312"},"uses_additional_data":true},{"leaderboard":"/sota/retrieval-augmented-few-shot-in-context-audio","task":"Retrieval-augmented Few-shot In-context Audio Captioning","dataset":"AudioCaps","model":"Audio Flamingo (4-shot)","rank_in_archive_order":1,"of":5,"metrics":{"CIDEr":"0.518"},"uses_additional_data":true},{"leaderboard":"/sota/zero-shot-audio-captioning-on-audiocaps","task":"Zero-shot Audio Captioning","dataset":"AudioCaps","model":"Audio Flamingo","rank_in_archive_order":1,"of":4,"metrics":{"BLEU-4":"14.3","CIDEr":"50.2","METEOR":"20.5","ROUGE-L":"40.8","SPICE":"15.1","SPIDEr":"32.6"},"uses_additional_data":true}],"syntology":{"syntology_url":null,"atlas_url":"https://app.syntology.ai/?focus=2402.01831","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01831"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"deterministic:regex_extraction","url":"https://github.com/NVIDIA/audio-flamingo","reach":{"status":"ok"}}],"summary":{"ran_violates":3},"by_repo_kind":{},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":3,"samples":[{"code_sha256_prefix":"856d86b39c6005b4","entry":"divisible_by","repo":null,"repo_kind":null,"path":null,"file_url":null,"link_basis":"identical_code_first_harvested_elsewhere","language":"python","status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"mcp_get_code":{"code_sha256":"856d86b39c6005b4"}},{"code_sha256_prefix":"60fff7c3c400d7ff","entry":"default","repo":null,"repo_kind":null,"path":null,"file_url":null,"link_basis":"identical_code_first_harvested_elsewhere","language":"python","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"mcp_get_code":{"code_sha256":"60fff7c3c400d7ff"}},{"code_sha256_prefix":"aa5486a3650902d8","entry":"exists","repo":null,"repo_kind":null,"path":null,"file_url":null,"link_basis":"identical_code_first_harvested_elsewhere","language":"python","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"mcp_get_code":{"code_sha256":"aa5486a3650902d8"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}