{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/soundnet-learning-sound-representations-from","title":"SoundNet: Learning Sound Representations from Unlabeled Video","arxiv_id":"1610.09001","date":"2016-10-27","proceeding":"NeurIPS 2016 12","authors":["Yusuf Aytar","Carl Vondrick","Antonio Torralba"],"abstract":"We learn rich natural sound representations by capitalizing on large amounts\nof unlabeled sound data collected in the wild. We leverage the natural\nsynchronization between vision and sound to learn an acoustic representation\nusing two-million unlabeled videos. Unlabeled video has the advantage that it\ncan be economically acquired at massive scales, yet contains useful signals\nabout natural sound. We propose a student-teacher training procedure which\ntransfers discriminative visual knowledge from well established visual\nrecognition models into the sound modality using unlabeled video as a bridge.\nOur sound representation yields significant performance improvements over the\nstate-of-the-art results on standard benchmarks for acoustic scene/object\nclassification. Visualizations suggest some high-level semantics automatically\nemerge in the sound network, even though it is trained without ground truth\nlabels.","url_abs":"http://arxiv.org/abs/1610.09001v1","url_pdf":"http://arxiv.org/pdf/1610.09001v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"soundnet-learning-sound-representations-from","repo_url":"https://github.com/Alexyuda/action_recognition","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"soundnet-learning-sound-representations-from","repo_url":"https://github.com/Nnra/AIAproject","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"unanswered"}},{"paper_slug":"soundnet-learning-sound-representations-from","repo_url":"https://github.com/atsiami/STAViS","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"soundnet-learning-sound-representations-from","repo_url":"https://github.com/cvondrick/soundnet","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"torch","reach":{"status":"unanswered"}},{"paper_slug":"soundnet-learning-sound-representations-from","repo_url":"https://github.com/eborboihuc/SoundNet-tensorflow","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"unanswered"}},{"paper_slug":"soundnet-learning-sound-representations-from","repo_url":"https://github.com/pseeth/soundnet_keras","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"tf","reach":{"status":"unanswered"}}],"tasks":[{"task_slug":"classification","task_name":"General Classification"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1610.09001","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}