{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/the-sjtu-system-for-dcase2021-challenge-task","title":"THE SJTU SYSTEM FOR DCASE2021 CHALLENGE TASK 6: AUDIO CAPTIONING BASED ON ENCODER PRE-TRAINING AND REINFORCEMENT LEARNING","arxiv_id":null,"date":"2021-07-06","proceeding":"DCASE Challenge 2021 7","authors":["Xuenan Xu","Zeyu Xie","Mengyue Wu","Kai Yu"],"abstract":"This report proposes an audio captioning system for the Detection\r\nand Classification of Acoustic Scenes and Events (DCASE) 2021\r\nchallenge task Task 6. Our audio captioning system consists of a\r\n10-layer convolution neural network (CNN) encoder and a tempo-\r\nral attentional single layer gated recurrent unit (GRU) decoder. In\r\nthis challenge, there is no restriction on the usage of external data\r\nand pre-trained models. To better model the concepts in an audio\r\nclip, we pre-train the CNN encoder with audio tagging on AudioSet.\r\nAfter standard cross entropy based training, we further fine-tune the\r\nmodel with reinforcement learning to directly optimize the evalua-\r\ntion metric. Experiments show that our proposed system achieves a\r\nSPIDEr of 28.6 on the public evaluation split without ensemble1.","url_abs":"https://dcase.community/documents/challenge2021/technical_reports/DCASE2021_Xu_119_t6.pdf","url_pdf":"https://dcase.community/documents/challenge2021/technical_reports/DCASE2021_Xu_119_t6.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"the-sjtu-system-for-dcase2021-challenge-task","repo_url":"https://github.com/wsntxxn/AudioCaption","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"audio-tagging","task_name":"Audio Tagging"},{"task_slug":"audio-captioning","task_name":"Audio captioning"},{"task_slug":"decoder","task_name":"Decoder"},{"task_slug":"reinforcement-learning-1","task_name":"Reinforcement Learning (RL)"},{"task_slug":"reinforcement-learning-2","task_name":"reinforcement-learning"}],"methods":[{"method_slug":"convolution","method_name":"Convolution"}],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/audio-captioning-on-clotho","task":"Audio captioning","dataset":"Clotho","model":"Ensemble-RL","rank_in_archive_order":6,"of":11,"metrics":{"CIDEr":"0.468","SPICE":"0.123","SPIDEr":"0.295"},"uses_additional_data":true}],"syntology":{"atlas_url":null,"mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}