{"url":"/sota/audio-captioning-on-clotho","task":{"name":"Audio captioning","url":"/task/audio-captioning","note":null},"dataset":{"name":"Clotho","url":"/dataset/clotho"},"category":"Audio","categories":["Audio"],"category_note":null,"description":"Audio Captioning is the task of describing audio using text. The general approach is to use an audio encoder to encode the audio (example: PANN, CAV-MAE), and to use a decoder (example: transformer) to generate the text.\r\nTo judge the quality of audio captions, though machine translation metrics (BLEU, METEOR, ROUGE)  and image captioning metrics (SPICE, CIDER) are used, they are not very well-suited. Attempts have been made to use pretrained language model based metrics such as Sentence-BERT.","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["SPIDEr","CIDEr","SPICE","BLEU-4","METEOR","ROUGE-L","FENSE","SPIDEr-FL","Sentence-BERT"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"SPIDEr":null,"CIDEr":null,"SPICE":null,"BLEU-4":"higher","METEOR":null,"ROUGE-L":"higher","FENSE":null,"SPIDEr-FL":null,"Sentence-BERT":null}},"counts":{"rows":11,"rows_with_code":7,"rows_with_paper_page":11,"rows_dated":11,"rows_using_additional_data":8},"rows":[{"rank_in_archive_order":1,"model":"SLAM-AAC","metrics":{"CIDEr":"0.515","FENSE":"0.540","METEOR":"0.197","SPICE":"0.148","SPIDEr":"0.332","SPIDEr-FL":"0.330"},"uses_additional_data":true,"paper_date":"2024-10-12","paper":"/paper/slam-aac-enhancing-audio-captioning-with","paper_url":"https://arxiv.org/abs/2410.09503v1","paper_title":"SLAM-AAC: Enhancing Audio Captioning with Paraphrasing Augmentation and CLAP-Refine through LLMs","code":"https://github.com/X-LANCE/SLAM-LLM","n_code_links":1,"syntology":null},{"rank_in_archive_order":2,"model":"LOAE","metrics":{"CIDEr":"0.513","FENSE":"0.538","METEOR":"0.197","SPICE":"0.147","SPIDEr":"0.330","SPIDEr-FL":"0.330","Sentence-BERT":"0.538"},"uses_additional_data":true,"paper_date":"2024-06-19","paper":"/paper/enhancing-automated-audio-captioning-via","paper_url":"https://arxiv.org/abs/2406.13275v2","paper_title":"Enhancing Automated Audio Captioning via Large Language Models with Optimized Audio Encoding","code":"https://github.com/frankenliu/LOAE","n_code_links":1,"syntology":{"n_ran":5,"n_unverified":3,"n_samples":8,"n_pointer_only_licence":0}},{"rank_in_archive_order":3,"model":"MQ-Cap","metrics":{"BLEU-4":"18.1","CIDEr":"0.496","METEOR":"0.192","SPICE":"0.143","SPIDEr":"0.319"},"uses_additional_data":false,"paper_date":"2024-10-14","paper":"/paper/audio-captioning-via-generative-pair-to-pair","paper_url":"https://arxiv.org/abs/2410.10913v3","paper_title":"Enhancing Retrieval-Augmented Audio Captioning with Generation-Assisted Multimodal Querying and Progressive Learning","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":4,"model":"Ensemble","metrics":{"CIDEr":"0.400","SPICE":"0.137","SPIDEr":"0.318"},"uses_additional_data":true,"paper_date":"2021-07-06","paper":"/paper/the-dcase-2021-challenge-task-6-system","paper_url":"https://dcase.community/documents/challenge2021/technical_reports/DCASE2021_Yuan_2_t6.pdf","paper_title":"THE DCASE 2021 CHALLENGE TASK 6 SYSTEM: AUTOMATED AUDIO CAPTIONING WITH WEAKLY SUPERVISED PRE-TRAING AND WORD SELECTION METHODS","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":5,"model":"Audio Flamingo (Pengi trainset)","metrics":{"BLEU-4":"17.4","CIDEr":"0.489","METEOR":"18.7","ROUGE-L":"39.4","SPICE":"0.134","SPIDEr":"0.312"},"uses_additional_data":true,"paper_date":"2024-02-02","paper":"/paper/audio-flamingo-a-novel-audio-language-model","paper_url":"https://arxiv.org/abs/2402.01831v3","paper_title":"Audio Flamingo: A Novel Audio Language Model with Few-Shot Learning and Dialogue Abilities","code":"https://github.com/NVIDIA/audio-flamingo","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":0,"n_samples":3,"n_pointer_only_licence":3}},{"rank_in_archive_order":6,"model":"Ensemble-RL","metrics":{"CIDEr":"0.468","SPICE":"0.123","SPIDEr":"0.295"},"uses_additional_data":true,"paper_date":"2021-07-06","paper":"/paper/the-sjtu-system-for-dcase2021-challenge-task","paper_url":"https://dcase.community/documents/challenge2021/technical_reports/DCASE2021_Xu_119_t6.pdf","paper_title":"THE SJTU SYSTEM FOR DCASE2021 CHALLENGE TASK 6: AUDIO CAPTIONING BASED ON ENCODER PRE-TRAINING AND REINFORCEMENT LEARNING","code":"https://github.com/wsntxxn/AudioCaption","n_code_links":1,"syntology":null},{"rank_in_archive_order":7,"model":"Qwen-Audio","metrics":{"CIDEr":"0.441","SPICE":"0.136","SPIDEr":"0.288"},"uses_additional_data":true,"paper_date":"2023-11-14","paper":"/paper/qwen-audio-advancing-universal-audio","paper_url":"https://arxiv.org/abs/2311.07919v2","paper_title":"Qwen-Audio: Advancing Universal Audio Understanding via Unified Large-Scale Audio-Language Models","code":"https://github.com/alibaba-damo-academy/FunASR","n_code_links":2,"syntology":{"n_ran":5,"n_unverified":2,"n_samples":7,"n_pointer_only_licence":7}},{"rank_in_archive_order":8,"model":"Ensemble","metrics":{"CIDEr":"0.319","SPICE":"0.094","SPIDEr":"0.207"},"uses_additional_data":false,"paper_date":"2020-07-01","paper":"/paper/the-ntt-dcase2020-challenge-task-6-system","paper_url":"https://arxiv.org/abs/2007.00225v1","paper_title":"The NTT DCASE2020 Challenge Task 6 system: Automated Audio Captioning with Keywords and Sentence Length Estimation","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":9,"model":"VAST","metrics":{"BLEU-4":"19","CIDEr":"0.519","METEOR":"19.3","ROUGE-L":"40.8"},"uses_additional_data":true,"paper_date":"2023-05-29","paper":"/paper/vast-a-vision-audio-subtitle-text-omni-1","paper_url":"https://arxiv.org/abs/2305.18500v2","paper_title":"VAST: A Vision-Audio-Subtitle-Text Omni-Modality Foundation Model and Dataset","code":"https://github.com/TXH-mercury/VALOR","n_code_links":2,"syntology":{"n_ran":15,"n_unverified":27,"n_samples":42,"n_pointer_only_licence":0}},{"rank_in_archive_order":10,"model":"VALOR","metrics":{"BLEU-4":"16.2","CIDEr":"0.423","METEOR":"17.4","ROUGE-L":"38.2"},"uses_additional_data":true,"paper_date":"2023-04-17","paper":"/paper/valor-vision-audio-language-omni-perception","paper_url":"https://arxiv.org/abs/2304.08345v2","paper_title":"VALOR: Vision-Audio-Language Omni-Perception Pretraining Model and Dataset","code":"https://github.com/TXH-mercury/VALOR","n_code_links":1,"syntology":null},{"rank_in_archive_order":11,"model":"RNN-GRU-EncDec + VGGish + Word2Vec","metrics":{"CIDEr":"0.18"},"uses_additional_data":false,"paper_date":"2020-06-05","paper":"/paper/audio-captioning-using-gated-recurrent-units","paper_url":"https://arxiv.org/abs/2006.03391v3","paper_title":"Audio Captioning using Gated Recurrent Units","code":null,"n_code_links":0,"syntology":null}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":4,"rows_with_any_sample_ran":4,"distinct_papers_with_graph_line":4,"distinct_papers_with_any_sample_ran":4,"samples_over_distinct_papers":{"n_ran":28,"n_unverified":32,"n_samples":60,"n_pointer_only_licence":10,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":28,"n_unverified":32,"n_samples":60,"n_pointer_only_licence":10,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}