{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/classifying-variable-length-audio-files-with","title":"Classifying Variable-Length Audio Files with All-Convolutional Networks and Masked Global Pooling","arxiv_id":"1607.02857","date":"2016-07-11","proceeding":null,"authors":["Lars Hertel","Huy Phan","Alfred Mertins"],"abstract":"We trained a deep all-convolutional neural network with masked global pooling\nto perform single-label classification for acoustic scene classification and\nmulti-label classification for domestic audio tagging in the DCASE-2016\ncontest. Our network achieved an average accuracy of 84.5% on the four-fold\ncross-validation for acoustic scene recognition, compared to the provided\nbaseline of 72.5%, and an average equal error rate of 0.17 for domestic audio\ntagging, compared to the baseline of 0.21. The network therefore improves the\nbaselines by a relative amount of 17% and 19%, respectively. The network only\nconsists of convolutional layers to extract features from the short-time\nFourier transform and one global pooling layer to combine those features. It\nparticularly possesses neither fully-connected layers, besides the\nfully-connected output layer, nor dropout layers.","url_abs":"http://arxiv.org/abs/1607.02857v1","url_pdf":"http://arxiv.org/pdf/1607.02857v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"classifying-variable-length-audio-files-with","repo_url":"https://github.com/numpde/phonepad","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null}],"tasks":[{"task_slug":"acoustic-scene-classification","task_name":"Acoustic Scene Classification"},{"task_slug":"all","task_name":"All"},{"task_slug":"audio-tagging","task_name":"Audio Tagging"},{"task_slug":"classification-1","task_name":"Classification"},{"task_slug":"classification","task_name":"General Classification"},{"task_slug":"multi-label-classification-2","task_name":"MUlTI-LABEL-ClASSIFICATION"},{"task_slug":"multi-label-classification","task_name":"Multi-Label Classification"},{"task_slug":"scene-classification","task_name":"Scene Classification"},{"task_slug":"scene-recognition","task_name":"Scene Recognition"}],"methods":[{"method_slug":"dropout","method_name":"Dropout"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":null,"mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}